igneum-pow at release-0.3.11 (d85f4de): the fork at 89dfcb95 builds against the Counter ASIC 2.0 crate (ledger close round 3)

The fork's consensus/pow and igneum/miner depend on `../../../../igneum-pow`, which resolves lexically to the checkout the
fork's worktree sits under. The 0.3.11 fork tip 89dfcb95 needs ProgramClass, Epoch::chain_program,
Epoch::chain_dataset_day and verify::days_since_genesis, none of which fud-close's copy (or master's) carries, so the
rebased fork `ledger-fixes-0311` cannot build from under the main checkout (9 errors in kaspa-pow, build-merge.log). This
worktree takes the crate as release-0.3.11 ships it, unchanged, and the fork worktree moves under
igneum-wt-ledger-rebase/vendor/ to build. Coupling for the closer: the next cut merges fud-close with release-0.3.11, where
this directory is already identical.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-labs 2026-10-05 23:46:14 +00:00
parent 09c1de48c0
commit b956f0f85e
11 changed files with 5453 additions and 235 deletions

View file

@ -11,12 +11,14 @@
//! | (c) dynamic | the program is interpreted for [`ACCEPT_UNITS`] (64) units of 32 lanes at base nonces drawn from SplitMix64 seeded with `FNV-1a-64("igneum-accept/" \|\| seed words as little-endian bytes)`, each `low32(next()) AND NOT 31`, with init words equal to the seed words and the closed-form dataset `dataset_elem(idx, S[0], S[1])` at [`ACCEPT_DATASET_LOG2`] (2^28 words) in place of the memory-hard dataset. Over the 2,048 evaluations: no register has a bit equal in every final value; no load site (iteration, instruction) reads one address in all 32 lanes of any unit; fewer than [`MAX_SATURATED`] (164, 1 percent of 16,384) final register values are 0 or 2^32 - 1; every output bit's ones count is within [`BIAS_TOLERANCE`] (136, 6 sigma) of 1,024; the distinct masked addresses read by one lane in one evaluation, summed over the 2,048 evaluations, exceed [`MIN_DISTINCT_SUM`] (245,760, a mean above 120 of the 128 loads) |
//!
//! The dynamic test uses the closed form so that it is a pure function of the program (no cache, no day) and
//! costs about a millisecond on one core. The census (section 7.3) checked on 100,000 programs that the
//! costs about a millisecond on one core. A hot-table load (`docs/plans/hot-table.md`) reads the closed form keyed by
//! seed words 2 and 3 at its multiply-shift index, a second pure table beside the dataset stand-in (words 0 and 1). The census (section 7.3) checked on 100,000 programs that the
//! closed-form verdict agrees with the memory-hard one on all but 39 threshold-edge cases.
use crate::generator::{Instr, Op, Program, INSTR_COUNT, ITERATIONS, LANES};
use crate::seed::{fnv1a64, SplitMix64};
use crate::verify::{dataset_elem, splitmix32};
use crate::memhard::hot_index;
use crate::verify::{dataset_elem, fold_words, load_index, splitmix32, ScratchModel};
/// Units (32-lane warps) the dynamic test interprets.
pub const ACCEPT_UNITS: usize = 64;
@ -33,6 +35,14 @@ pub const BIAS_TOLERANCE: u32 = 136;
/// Distinct addresses per lane per evaluation, summed over 2,048 evaluations, must exceed this (mean above 120).
pub const MIN_DISTINCT_SUM: u64 = 245_760;
/// The distinct-address bound for a program with `loads` dataset loads per hash: the same 120 of 128 ratio, so
/// [`MIN_DISTINCT_SUM`] for the lottery hash and `loads x 1,920` for the read-width classes with other counts.
/// Variant 5's scratch read-modify-writes are not dataset loads: their slots repeat by design (a later
/// read-modify-write sees an earlier write), so they are neither counted nor bounded here.
pub fn min_distinct_sum(loads: usize) -> u64 {
loads as u64 * ACCEPT_HASHES as u64 * 120 / 128
}
/// Why a candidate was rejected. The verdict (accept or reject) is what consensus depends on; the reason is the
/// first failing test in the order of the module table.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
@ -67,7 +77,7 @@ impl std::fmt::Display for Reject {
Reject::Saturated { count } => write!(f, "(c) {count} of 16384 final register values saturated (limit 163)"),
Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"),
Reject::DistinctAddresses { sum } => {
write!(f, "(c) distinct addresses {sum} over 2048 hashes (mean {:.2}, needs above 120)", *sum as f64 / 2048.0)
write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0)
}
}
}
@ -170,6 +180,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
let seed = &p.seed;
let mask: u32 = (1u32 << ACCEPT_DATASET_LOG2) - 1;
let (d0, d1) = (seed[0], seed[1]);
let (h0, h1) = (seed[2], seed[3]);
let hot_words = p.hot_words();
let loads = p.loads_per_hash();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
@ -183,12 +195,30 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
}
let mut idx = [0u32; LANES];
let mut nload = 0usize;
let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None };
let slot_mask = p.class.scratch_slot_mask();
let era = p.class.era;
for it in 0..ITERATIONS {
let sel = r[0];
for (k, ins) in p.instrs.iter().enumerate() {
let d = ins.dst as usize;
let a = ins.src as usize;
match ins.op {
Op::Scratch => {
// Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word).
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
for lane in 0..LANES {
idx[lane] = r[a][lane] & slot_mask;
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
r[d][lane] = m.rmw(&p.seed, base, lane, idx[lane], r[d][lane]);
lane_addrs[lane * loads + nload] = 0x8000_0000 | idx[lane];
}
nload += 1;
}
Op::Add => {
let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32);
let src = r[a];
@ -255,18 +285,45 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
}
}
Op::Load => {
// Read-width experiment: a load of `width` words reads from the aligned address and folds every
// word (verify::fold_words); width 1 is the lottery hash's xor of one word.
let width = ins.width as usize;
let align = !(ins.width as u32 - 1);
for lane in 0..LANES {
idx[lane] = r[a][lane] & mask;
idx[lane] = load_index(era.as_ref(), ins, r[a][lane], mask, ACCEPT_DATASET_LOG2) & align;
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
r[d][lane] ^= dataset_elem(idx[lane], d0, d1);
if width == 1 {
r[d][lane] ^= dataset_elem(idx[lane], d0, d1);
} else {
let mut w = [0u32; 16];
for j in 0..width {
w[j] = dataset_elem(idx[lane] + j as u32, d0, d1);
}
r[d][lane] = fold_words(r[d][lane], &w[..width]);
}
lane_addrs[lane * loads + nload] = idx[lane];
}
nload += 1;
}
Op::Hot => {
// Hot table: the stand-in is dataset_elem keyed by seed words 2 and 3; the address is tagged with
// bit 30 so a hot word and a dataset word at one index count as two addresses.
for lane in 0..LANES {
idx[lane] = hot_index(r[a][lane], hot_words);
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
r[d][lane] ^= dataset_elem(idx[lane], h0, h1);
lane_addrs[lane * loads + nload] = 0x4000_0000 | idx[lane];
}
nload += 1;
}
Op::WLoad => {
let b = (r[a][0] & mask) & !31;
for lane in 0..LANES {
@ -298,7 +355,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
sl.sort_unstable();
let mut distinct = 0u64;
for k in 0..loads {
if k == 0 || sl[k] != sl[k - 1] {
// scratch slots carry bit 31 (variant 5) and are not dataset addresses
if sl[k] & 0x8000_0000 == 0 && (k == 0 || sl[k] != sl[k - 1]) {
distinct += 1;
}
}
@ -333,7 +391,7 @@ pub fn check_dynamic(p: &Program) -> Result<AcceptReport, Reject> {
}
bias_max = bias_max.max(d);
}
if acc.distinct_sum <= MIN_DISTINCT_SUM {
if acc.distinct_sum <= min_distinct_sum(loads - p.scratch_ops_per_hash()) {
return Err(Reject::DistinctAddresses { sum: acc.distinct_sum });
}
Ok(AcceptReport { distinct_sum: acc.distinct_sum, saturated: acc.saturated, bias_max })
@ -348,9 +406,50 @@ pub fn check(p: &Program) -> Result<AcceptReport, Reject> {
#[cfg(test)]
mod tests {
use super::*;
use crate::generator::{candidate, generate, GeneratorConfig, generate_v1};
use crate::generator::{candidate, candidate_class, generate, generate_class, GeneratorConfig, generate_v1, LoadClass};
use crate::verify::{DatasetMode, DatasetSource};
#[test]
fn distinct_bound_scales_with_the_load_count() {
assert_eq!(min_distinct_sum(128), MIN_DISTINCT_SUM);
assert_eq!(min_distinct_sum(32), 61_440);
}
/// The read-width classes pass the rule at about the version 2 rate, and the instrumented interpreter agrees
/// with `verify.rs` on every class (the fold is shared, the addresses are aligned the same way).
#[test]
fn classes_pass_and_match_verify() {
for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2k32", "scr8k128"] {
let c = LoadClass::parse(name).unwrap();
let p = generate_class("igneum-genesis", c);
assert!(check(&p).is_ok(), "{name}");
let mut rejected = 0;
for i in 0..60u32 {
let s = format!("igneum-rw-accept/{i}");
let q = candidate_class(&s, s.as_bytes(), 0, c);
if check(&q).is_err() {
rejected += 1;
}
}
assert!(rejected < 15, "{name}: {rejected} of 60 rejected");
let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2);
let bases = accept_base_nonces(&p.seed);
let loads = p.loads_per_hash();
let mut acc = Acc { and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
let mut la = vec![0u32; LANES * loads];
let mut ones = [0u32; 64];
for (u, &b) in bases.iter().enumerate() {
run_unit(&p, u, b, &mut acc, &mut la).unwrap();
for h in crate::verify::hash_warp(&p, b, &ds) {
for j in 0..64 {
ones[j] += ((h >> j) & 1) as u32;
}
}
}
assert_eq!(acc.bit_ones, ones, "{name}: bit counts match the reference interpreter");
}
}
/// The instrumented interpreter agrees with `verify.rs` on the closed-form dataset keyed by the seed words.
#[test]
fn instrumented_interpreter_matches_verify() {
@ -381,6 +480,52 @@ mod tests {
}
}
/// Hot-table experiment: the hot classes pass the rule at about the version 2 rate, and the hot addresses are
/// uniform over the table (16 buckets of the index over 64 units x 32 lanes x 32 hot loads).
#[test]
fn hot_classes_pass_and_hot_loads_are_uniform() {
for name in ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "scr4k32+hot64k4", "hot32k4a", "hot64k4a", "hot96k4a"] {
let c = LoadClass::parse(name).unwrap();
let p = generate_class("igneum-genesis", c);
assert!(check(&p).is_ok(), "{name}");
let mut rejected = 0;
for i in 0..60u32 {
let s = format!("igneum-hot-accept/{i}");
let q = candidate_class(&s, s.as_bytes(), 0, c);
if check(&q).is_err() {
rejected += 1;
}
}
assert!(rejected < 15, "{name}: {rejected} of 60 rejected");
}
let p = generate_class("igneum-genesis", LoadClass::hot(96, 4));
let words = p.hot_words();
let loads = p.loads_per_hash();
let mut acc = Acc { and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
let mut la = vec![0u32; LANES * loads];
let mut buckets = [0u64; 16];
let mut hot_count = 0u64;
for (u, &b) in accept_base_nonces(&p.seed).iter().enumerate() {
run_unit(&p, u, b, &mut acc, &mut la).unwrap();
for &a in &la {
if a & 0xC000_0000 == 0x4000_0000 {
let idx = a & 0x3FFF_FFFF;
assert!(idx < words);
buckets[(idx as u64 * 16 / words as u64) as usize] += 1;
hot_count += 1;
}
}
}
assert_eq!(hot_count, 64 * 32 * 32, "32 hot loads per hash over 2,048 hashes");
let mean = hot_count as f64 / 16.0;
for (i, &b) in buckets.iter().enumerate() {
assert!((b as f64 - mean).abs() < 0.15 * mean, "bucket {i}: {b} against a mean of {mean}");
}
// the dataset distinct count still holds for the dataset loads alone
let r = check(&p).unwrap();
assert!(r.distinct_mean() > 120.0);
}
#[test]
fn base_nonces_are_aligned_and_seed_dependent() {
let a = accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 8]);

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

View file

@ -26,12 +26,13 @@ pub mod bind;
pub mod emit;
pub mod generator;
pub mod memhard;
pub mod packcheck;
pub mod seed;
pub mod verify;
pub use bind::{block_init_words, day_bytes, pow256_from_lane, target64_from_le256};
pub use accept::{check as accept_program, AcceptReport, Reject};
pub use generator::{generate, generate_from_seed_bytes, Instr, Op, Program, GENERATOR_VERSION};
pub use memhard::{Cache, MemhardCpu, MixParams};
pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, V3_CLASS};
pub use memhard::{cache_log2_words, dataset_log2_words, days_since_genesis, growth_doublings, Cache, MemhardCpu, MixParams, Shape};
pub use seed::{fnv1a64, seed_words, SplitMix64};
pub use verify::{hash_warp, interpret_warp_init, verify_block, DatasetMode, DatasetSource, Epoch};

View file

@ -7,11 +7,22 @@
//! [--epoch-hex <64 hex> --day-hex <hex>] byte seeds instead of strings (Epoch::from_seed_bytes)
//! igneum-pow accept --seed <s> [--epoch-hex <64 hex>] every candidate of the seed with its verdict (spec 01 section 1.4.6)
//! igneum-pow show --seed <s> [--epoch-hex <64 hex>] the accepted program, one instruction per line
//!
//! Read-width experiment (5 October 2026, docs/plans/read-width.md): `--class v2|w4|w16|w64|w64x4|p4,p16,p64[xN]`
//! on every command selects the load class (default v2, the lottery hash). Nothing in a v2 run changes.
//!
//! Era layout (5 October 2026, docs/plans/era-layout.md): `--era igneum-era-test/<n>` (a test era seed: the 32 bytes
//! of seed_words_from_bytes of the string, era index n) or `--era <n>:<64 hex>` (the chain's 32-byte era seed E_n)
//! turns the chosen class into its era class; `--era-widths 4` (default: the read-width decision of 5 October 2026
//! keeps v2's 4-byte load) is the allowed width set the era draws from; `4,16,64` lets the era draw the width.
use igneum_pow::generator::{EraParams, GENERATOR_VERSION_V3};
use igneum_pow::emit::export_pack;
use igneum_pow::memhard::Cache;
use igneum_pow::generator::{LoadClass, ProgramClass};
use igneum_pow::memhard::{Cache, Shape};
use igneum_pow::seed::day_key;
use igneum_pow::verify::{DatasetMode, Epoch, DEFAULT_DATASET_LOG2};
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2};
use std::time::Instant;
struct Args {
@ -26,6 +37,51 @@ struct Args {
prehash: String,
epoch_hex: Option<String>,
day_hex: Option<String>,
class: LoadClass,
/// Days since genesis for the cache growth rule of a class with `growth` (0: the genesis cache).
days: u64,
/// The program class (Counter ASIC 2.0 seam): v2 (default) or v3, which draws from V3_CLASS with generator 3.
program_class: Option<ProgramClass>,
/// The era seed bytes a class v3 chain program records (`--era-hex`).
era_hex: Option<String>,
/// Era layout: `--era igneum-era-test/<n>` or `--era <n>:<64 hex>` composes the era class over `--class` with
/// generator 3 and the era bytes recorded (the measurement packs: v2's mixer under the era layout).
era: Option<(u64, Vec<u8>, String)>,
era_widths: Vec<u8>,
}
/// `igneum-era-test/<n>` or `<n>:<64 hex>` -> (index, 32 era bytes, label).
fn parse_era(s: &str) -> Option<(u64, Vec<u8>, String)> {
if let Some(n) = s.strip_prefix("igneum-era-test/") {
let index: u64 = n.parse().ok()?;
return Some((index, EraParams::test_era_bytes(s).to_vec(), s.to_string()));
}
let (n, hex) = s.split_once(':')?;
let index: u64 = n.parse().ok()?;
let bytes = igneum_pow::bind::unhex(hex)?;
if bytes.len() != 32 {
return None;
}
Some((index, bytes, format!("igneum-era/{index}/{hex}")))
}
/// "4,16,64" (bytes) -> ascending words.
fn parse_widths(s: &str) -> Option<Vec<u8>> {
let mut v: Vec<u8> = s
.split(',')
.map(|x| match x.trim() {
"4" => Some(1u8),
"16" => Some(4),
"64" => Some(16),
_ => None,
})
.collect::<Option<Vec<_>>>()?;
v.sort_unstable();
v.dedup();
if v.is_empty() {
return None;
}
Some(v)
}
fn usage() -> ! {
@ -36,7 +92,12 @@ fn usage() -> ! {
\x20 hash --nonce <n> print the 64-bit hash of one nonce (pack form, init words = seed words)\n\
\x20 hash-bound --prehash <64 hex> --nonce <u64> print the header-bound hash (bind.rs) of one 64-bit nonce\n\
\x20 accept every candidate of the seed (or --epoch-hex) with its acceptance verdict\n\
\x20 show the accepted program, one instruction per line"
\x20 show the accepted program, one instruction per line\n\
\x20 --class C load class: v2 (default), mx4 (class v3: mixer x4, cache growth), w4, w16, w64, w64x4, p4,p16,p64[xN], <class>m<mult>[g]\n\
\x20 --days N days since genesis for the cache growth rule of a class with it (default 0: the 2^26-word cache)\n\
\x20 --program-class v2|v3 the program class of the seam (v3 = generator 3 on V3_CLASS, the chain's own derivation; --era-hex records the era seed)\n\
\x20 --era E era layout over --class: igneum-era-test/<n> or <n>:<64 hex> (the 32-byte era seed E_n)\n\
\x20 --era-widths 4[,16,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it)"
);
std::process::exit(2)
}
@ -54,6 +115,12 @@ fn parse() -> Args {
prehash: "00".repeat(32),
epoch_hex: None,
day_hex: None,
class: LoadClass::V2,
days: 0,
program_class: None,
era_hex: None,
era: None,
era_widths: vec![1],
};
let mut it = std::env::args().skip(1);
a.cmd = it.next().unwrap_or_else(|| usage());
@ -70,12 +137,30 @@ fn parse() -> Args {
"--prehash" => a.prehash = val(),
"--epoch-hex" => a.epoch_hex = Some(val()),
"--day-hex" => a.day_hex = Some(val()),
"--class" => a.class = LoadClass::parse(&val()).unwrap_or_else(|| usage()),
"--days" => a.days = val().parse().unwrap_or_else(|_| usage()),
"--program-class" => a.program_class = Some(ProgramClass::parse(&val()).unwrap_or_else(|| usage())),
"--era-hex" => a.era_hex = Some(val()),
"--era" => a.era = Some(parse_era(&val()).unwrap_or_else(|| usage())),
"--era-widths" => a.era_widths = parse_widths(&val()).unwrap_or_else(|| usage()),
_ => usage(),
}
}
if let Some((_, bytes, _)) = &a.era {
a.class = LoadClass::era(a.class, bytes, &a.era_widths);
}
a
}
/// An era program is a class v3 program: generator 3 and the era bytes recorded (what the chain's
/// `Epoch::from_chain_seeds` does); the pack then carries IGNEUM_PROGRAM_CLASS "v3" and IGNEUM_ERA_SEED_HEX.
fn stamp_era(e: &mut Epoch, a: &Args) {
if let Some((_, bytes, _)) = &a.era {
e.program.generator = GENERATOR_VERSION_V3;
e.program.era_bytes = Some(bytes.clone());
}
}
fn main() {
let a = parse();
let mode = if a.closed_form { DatasetMode::ClosedForm } else { DatasetMode::MemoryHard };
@ -85,21 +170,14 @@ fn main() {
"accept" => accept(&a),
"show" => show(&a),
"hash" => {
let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2);
let (e, _) = epoch_of(&a, mode);
println!("{:016x}", e.hash(a.nonce as u32));
}
"hash-bound" => {
let bytes = igneum_pow::bind::unhex(&a.prehash).unwrap_or_else(|| usage());
let prehash: [u8; 32] = bytes.as_slice().try_into().unwrap_or_else(|_| usage());
// --epoch-hex / --day-hex: the chain's byte seeds (Epoch::from_seed_bytes), as the worker protocol carries them
let e = match (&a.epoch_hex, &a.day_hex) {
(Some(eh), Some(dh)) => {
let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage());
let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage());
Epoch::from_seed_bytes(&eb, &db, "cli")
}
_ => Epoch::new(&a.seed, &a.day, mode, a.dataset_log2),
};
let (e, _) = epoch_of(&a, mode);
let init = igneum_pow::bind::block_init_words(&prehash, a.nonce);
println!("init words {}", init.iter().map(|w| format!("{w:08x}")).collect::<Vec<_>>().join(" "));
println!("{:016x}", e.hash_bound(&prehash, a.nonce));
@ -108,6 +186,49 @@ fn main() {
}
}
/// The epoch every command works on, and the day label for packs. `--epoch-hex`/`--day-hex` give the chain's byte
/// seeds (the day label then names the day bytes); else the string seed and day. `--program-class v3` draws the
/// program through the seam (generator 3 on `V3_CLASS`, the era bytes of `--era-hex` recorded) and sizes the
/// dataset for `--days` through `Epoch::chain_dataset_day`; `--class` is ignored under a program class (the class
/// names the load class). Closed-form mode is only for string seeds under the default class.
fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) {
let (mut e, label) = epoch_of_class(a, mode);
stamp_era(&mut e, a);
(e, label)
}
fn epoch_of_class(a: &Args, mode: DatasetMode) -> (Epoch, String) {
let era = a.era_hex.as_ref().map(|h| igneum_pow::bind::unhex(h).unwrap_or_else(|| usage()));
match (&a.epoch_hex, &a.day_hex) {
(Some(eh), Some(dh)) => {
let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage());
let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage());
let label = format!("igneum-epoch/{eh}/day/{dh}");
let e = match a.program_class {
Some(pc) => Epoch {
program: Epoch::chain_program(&eb, era.as_deref(), pc, &label),
dataset: Epoch::chain_dataset_day(&db, pc, a.days, a.dataset_log2),
},
None => Epoch::from_seed_bytes_day(&eb, &db, &label, a.class, a.days, a.dataset_log2),
};
(e, format!("bytes:{dh}"))
}
_ => {
let e = match a.program_class {
Some(pc) => {
let program = igneum_pow::generator::generate_from_seed_bytes_program_class(&a.seed, a.seed.as_bytes(), pc, era.as_deref());
let lc = pc.load_class();
let shape = Shape::for_class_day(&lc, a.days);
let log2 = if lc.growth { igneum_pow::memhard::dataset_log2_words(a.dataset_log2, a.days) } else { a.dataset_log2 };
Epoch { program, dataset: DatasetSource::new_shape(&a.day, mode, log2, shape) }
}
None => Epoch::new_class_day(&a.seed, &a.day, mode, a.dataset_log2, a.class, a.days),
};
(e, a.day.clone())
}
}
}
fn bench(a: &Args, mode: DatasetMode) {
println!(
"igneum-pow bench: seed \"{}\", day \"{}\", dataset 2^{} words ({})",
@ -116,21 +237,41 @@ fn bench(a: &Args, mode: DatasetMode) {
a.dataset_log2,
mode.name()
);
let shape = Shape::for_class_day(&a.program_class.map(|pc| pc.load_class()).unwrap_or(a.class), a.days);
if mode == DatasetMode::MemoryHard {
// Time the cache fill on its own first (one core), then build the epoch (which fills it again).
let t0 = Instant::now();
let c = Cache::fill(day_key(&a.day));
let c = Cache::fill_log2(day_key(&a.day), shape.cache_log2_words);
let fill_ms = t0.elapsed().as_secs_f64() * 1e3;
println!("cache: fill {fill_ms:.1} ms on one core (2^26 words, 65536 chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", c.fnv1a64());
println!(
"cache: fill {fill_ms:.1} ms on one core (2^{} words, {} MiB, {} chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}",
shape.cache_log2_words,
shape.cache_words() * 4 / (1 << 20),
c.segments(),
c.fnv1a64()
);
drop(c);
}
if let Some(h) = a.class.hot {
// the hot table of the epoch on its own first (one core), then the epoch (which fills it again)
let t0 = Instant::now();
let t = igneum_pow::memhard::HotTable::for_seed_bytes(a.seed.as_bytes(), h.mb as u32);
let fill_ms = t0.elapsed().as_secs_f64() * 1e3;
println!("hot table: {} MiB filled in {fill_ms:.1} ms on one core ({} chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", h.mb, igneum_pow::memhard::hot_segments(h.mb as u32), t.fnv1a64());
}
let t0 = Instant::now();
let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2);
let (e, _) = epoch_of(a, mode);
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
println!(
"program: {} loads/hash, {} items/warp, op mix {}; epoch built in {build_ms:.1} ms",
"program: class {}, {} loads/hash, {} bytes/hash, widths (1,4,16 words) {:?}, {} items/warp, mixer x{} ({} mixers/item), cache 2^{} words, op mix {}; epoch built in {build_ms:.1} ms",
e.program.class.name(),
e.program.loads_per_hash(),
e.program.bytes_per_hash(),
e.program.width_counts(),
e.program.items_per_warp(),
shape.mixer_mult,
shape.mixers_per_item(),
shape.cache_log2_words,
e.program.op_mix()
);
let bases = [0u32, 4096, 1_000_000];
@ -158,14 +299,7 @@ fn export(a: &Args, mode: DatasetMode) {
let out = a.out.clone().unwrap_or_else(|| usage());
let t0 = Instant::now();
// --epoch-hex / --day-hex: the chain's byte seeds; the day label then names the day bytes
let (e, day_label) = match (&a.epoch_hex, &a.day_hex) {
(Some(eh), Some(dh)) => {
let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage());
let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage());
(Epoch::from_seed_bytes(&eb, &db, &format!("igneum-epoch/{eh}/day/{dh}")), format!("bytes:{dh}"))
}
_ => (Epoch::new(&a.seed, &a.day, mode, a.dataset_log2), a.day.clone()),
};
let (e, day_label) = epoch_of(a, mode);
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
println!("igneum-pow export {out}");
println!(
@ -179,7 +313,7 @@ fn export(a: &Args, mode: DatasetMode) {
e.program.program_id(),
e.program.loads_per_hash()
);
println!("op mix: {}", e.program.op_mix());
println!("op mix: {}; class {}, {} bytes/hash, widths (1,4,16 words) {:?}", e.program.op_mix(), e.program.class.name(), e.program.bytes_per_hash(), e.program.width_counts());
let source = format!("igneum-pow (Rust) CPU interpreter, generator v{}, {} dataset", e.program.generator, e.dataset.mode().name());
let pack = export_pack(&e, &day_label, &source);
let dir = std::path::Path::new(&out);
@ -209,7 +343,7 @@ fn seed_bytes_of(a: &Args) -> (String, Vec<u8>) {
fn accept(a: &Args) {
let (label, bytes) = seed_bytes_of(a);
let t0 = Instant::now();
let tries = igneum_pow::generator::attempts(&label, &bytes);
let tries = igneum_pow::generator::attempts_class(&label, &bytes, a.class);
let ms = t0.elapsed().as_secs_f64() * 1e3;
for (p, verdict) in &tries {
match verdict {
@ -233,19 +367,42 @@ fn accept(a: &Args) {
fn show(a: &Args) {
let (label, bytes) = seed_bytes_of(a);
let p = igneum_pow::generator::generate_from_seed_bytes(&label, &bytes);
let mut p = igneum_pow::generator::generate_from_seed_bytes_class(&label, &bytes, a.class);
if let Some((_, eb, _)) = &a.era {
p.generator = GENERATOR_VERSION_V3;
p.era_bytes = Some(eb.clone());
}
println!(
"seed \"{}\" generator v{} attempt {} program id {:016x} seed words {}",
"seed \"{}\" generator v{} class {} attempt {} program id {:016x} seed words {}",
p.seed_string,
p.generator,
p.class.name(),
p.attempt,
p.program_id(),
p.seed.iter().map(|w| format!("{w:08x}")).collect::<Vec<_>>().join(" ")
);
println!("op mix {} loads/hash {}", p.op_mix(), p.loads_per_hash());
println!("op mix {} loads/hash {} bytes/hash {}", p.op_mix(), p.loads_per_hash(), p.bytes_per_hash());
if let Some(e) = p.class.era {
println!(
"era {} ({}): width {} B, stride mul {:#010x} rot {}, interleave {:?}, windows (site:shrink:offset) {}",
e.label(),
a.era.as_ref().map(|x| x.2.as_str()).unwrap_or("?"),
e.width_words as u32 * 4,
e.stride_mul,
e.stride_rot,
e.pos,
p.instrs
.iter()
.enumerate()
.filter(|(_, i)| i.op == igneum_pow::generator::Op::Load)
.map(|(k, i)| format!("{k}:{}:{}", i.win, i.off))
.collect::<Vec<_>>()
.join(" ")
);
}
for (k, i) in p.instrs.iter().enumerate() {
println!(
"{k:2}: {:5} dst={} src={} src2={} imm={:#010x} imm2={:#010x} rot={} bit={} mask={}",
"{k:2}: {:5} dst={} src={} src2={} imm={:#010x} imm2={:#010x} rot={} bit={} mask={}{}",
i.op.name(),
i.dst,
i.src,
@ -254,7 +411,8 @@ fn show(a: &Args) {
i.imm2,
i.rot,
i.bit,
i.mask
i.mask,
if i.op == igneum_pow::generator::Op::Load && i.width > 1 { format!(" width={}B", i.width as u32 * 4) } else { String::new() }
);
}
}

View file

@ -3,12 +3,18 @@
//! ARX-multiply mixer. The verifier holds the cache and never the dataset.
//!
//! All arithmetic is on u32 modulo 2^32. Rotations are by 1..31 at every call site.
//!
//! Counter ASIC 2.0 (5 October 2026, `docs/plans/mixer-x4.md`, behind the program class): the construction has a
//! [`Shape`], the mixer multiplier `m` and the cache size. Under `m` every mixer application of an item becomes
//! `m` applications with distinct round keys, the 8 dependent cache reads unchanged; the cache doubles when the
//! dataset doubles ([`growth_doublings`]). [`Shape::V2`] (`m = 1`, 2^26 words) is version 2 bit for bit.
use crate::generator::LoadClass;
use crate::seed::{day_key, fnv1a64_words, SplitMix64};
pub const CACHE_LOG2_WORDS: usize = 26;
pub const CACHE_SEGMENT_LOG2_LINES: usize = 6;
/// 2^26 words = 256 MiB.
/// 2^26 words = 256 MiB (the version 2 cache, and the v3 cache until the first dataset doubling).
pub const CACHE_WORDS: usize = 1 << CACHE_LOG2_WORDS;
/// 2^22 lines of 16 words.
pub const CACHE_LINES: usize = CACHE_WORDS >> 4;
@ -23,6 +29,102 @@ pub const CHACHA_ROUNDS: usize = 12;
pub const CHACHA_SIGMA: [u32; 4] = [0x61707865, 0x3320646e, 0x79622d32, 0x6b206574];
/// "Igne", "umMH".
pub const CACHE_TAG: [u32; 2] = [0x49676e65, 0x756d4d48];
/// "Igne", "umHT": the chain tag of the hot table (hot-table experiment, `docs/plans/hot-table.md`).
pub const HOT_TAG: [u32; 2] = [0x49676e65, 0x756d4854];
/// Domain tag of the hot key: `KH = seed_words_from_bytes("igneum-hot/" || epoch seed bytes)`.
pub const HOT_KEY_TAG: &[u8] = b"igneum-hot/";
/// Words per MiB of hot table.
pub const HOT_WORDS_PER_MIB: u32 = 1 << 18;
/// Segments (64 chained lines of 16 words, 4 KiB) per MiB of hot table.
pub const HOT_SEGMENTS_PER_MIB: u32 = 256;
/// The shape of the item derivation and of the cache: the mixer multiplier and the cache size.
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
pub struct Shape {
/// Mixer applications per round (and after the last read): 1 under version 2, 4 under class v3.
pub mixer_mult: u32,
/// The cache is 2^cache_log2_words words (26 at genesis; 27 and 28 after the dataset doublings of 1.13.3).
pub cache_log2_words: u32,
}
impl Shape {
/// Version 2: one mixer application per round, a 2^26-word cache.
pub const V2: Shape = Shape { mixer_mult: 1, cache_log2_words: CACHE_LOG2_WORDS as u32 };
/// The shape of a load class on day 0 of the chain (and on every day for a class without the growth rule).
pub fn for_class(class: &LoadClass) -> Shape {
Shape::for_class_day(class, 0)
}
/// The shape of a load class on day `days_since_genesis` of the chain: the class's multiplier, and the cache
/// of [`cache_log2_words`] when the class has the growth rule, else 2^26 words.
pub fn for_class_day(class: &LoadClass, days_since_genesis: u64) -> Shape {
Shape {
mixer_mult: class.mixer_mult(),
cache_log2_words: if class.growth { cache_log2_words(days_since_genesis) } else { CACHE_LOG2_WORDS as u32 },
}
}
pub fn is_v2(&self) -> bool {
*self == Shape::V2
}
pub fn cache_words(&self) -> usize {
1usize << self.cache_log2_words
}
pub fn cache_lines(&self) -> usize {
self.cache_words() >> 4
}
pub fn cache_line_mask(&self) -> u32 {
(self.cache_lines() - 1) as u32
}
pub fn cache_segments(&self) -> usize {
self.cache_lines() >> CACHE_SEGMENT_LOG2_LINES
}
pub fn log2_segments(&self) -> u32 {
self.cache_log2_words - 4 - CACHE_SEGMENT_LOG2_LINES as u32
}
/// Mixer applications per item: `(ITEM_ROUNDS + 1) x m`.
pub fn mixers_per_item(&self) -> u32 {
(ITEM_ROUNDS as u32 + 1) * self.mixer_mult
}
}
// --------------------------------------------------------------------------------------------------------------
// Dataset growth, option C (spec 01 section 1.13.3 option (b) with the cache tied to the dataset's doublings)
// --------------------------------------------------------------------------------------------------------------
/// Days per year of the growth schedule: one year = 31,536,000 DAA seconds of 86,400 (spec 01 section 1.13.3).
pub const GROWTH_DAYS_PER_YEAR: u64 = 365;
/// The linear schedule of 1.13.3, 2 GiB at genesis plus 0.5 GiB per year, is `G x (1 + d / 1460)` for the genesis
/// size `G` and the day `d`: it doubles at day 1,460 (year 4), quadruples at day 4,380 (year 12), reaches 8x at
/// day 10,220 (year 28) and 16x at day 21,900 (year 60).
pub const GROWTH_DOUBLING_DAYS: u64 = 4 * GROWTH_DAYS_PER_YEAR;
/// The number of dataset doublings reached by day `days_since_genesis` of the chain: `floor(log2(1 + d / 1460))`,
/// in integers (`1 + d / 1460` rounded down, then its integer log2, which equals the real log2's floor because a
/// power of two is an integer). 0 until day 1,459; 1 from day 1,460 (year 4); 2 from day 4,380 (year 12).
pub fn growth_doublings(days_since_genesis: u64) -> u32 {
(1 + days_since_genesis / GROWTH_DOUBLING_DAYS).ilog2()
}
/// The cache size on day `d` under option C: 2^26 words doubled once per dataset doubling (256 MiB, 512 MiB from
/// year 4, 1 GiB from year 12).
pub fn cache_log2_words(days_since_genesis: u64) -> u32 {
CACHE_LOG2_WORDS as u32 + growth_doublings(days_since_genesis)
}
/// The dataset size on day `d` under option (b) of 1.13.3: the genesis size (2^`genesis_log2_words` words: 28 for
/// the 1 GiB packs and the devnet, 29 for the designed 2 GiB) doubled once per doubling of the linear schedule. The
/// result is capped at 32 (the item index is 32 bits, spec 1.13.3).
pub fn dataset_log2_words(genesis_log2_words: u32, days_since_genesis: u64) -> u32 {
(genesis_log2_words + growth_doublings(days_since_genesis)).min(32)
}
/// Days since genesis from two day indices of `bind::day_index` (the header's `timestamp_ms / 86,400,000`): the
/// day of the block and the day of the genesis header. A block before the genesis day (clock skew) is day 0.
pub fn days_since_genesis(day_index: u64, genesis_day_index: u64) -> u64 {
day_index.saturating_sub(genesis_day_index)
}
#[inline(always)]
fn rotl(x: u32, n: u32) -> u32 {
@ -66,17 +168,23 @@ pub fn chacha_block(x: &[u32; 16]) -> [u32; 16] {
y
}
/// Mixer parameters drawn from the day key. Draw order: ROT[0..7] (1..31), MUL[0..15] (odd), RC[0..15].
/// Mixer parameters drawn from the day key, plus the [`Shape`] the mixer is applied under. Draw order:
/// ROT[0..7] (1..31), MUL[0..15] (odd), RC[0..15]. The shape is not drawn: it is the class's.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct MixParams {
pub key: [u32; 8],
pub rot: [u32; 8],
pub mul: [u32; 16],
pub rc: [u32; 16],
pub shape: Shape,
}
impl MixParams {
/// Version 2 shape.
pub fn new(key: [u32; 8]) -> Self {
Self::with_shape(key, Shape::V2)
}
pub fn with_shape(key: [u32; 8], shape: Shape) -> Self {
let mut rng = SplitMix64::new(key[0] as u64 | ((key[1] as u64) << 32));
let mut rot = [0u32; 8];
let mut mul = [0u32; 16];
@ -90,7 +198,7 @@ impl MixParams {
for c in rc.iter_mut() {
*c = rng.next() as u32;
}
Self { key, rot, mul, rc }
Self { key, rot, mul, rc, shape }
}
/// Parameters for a day string: the key is `seed_words("day/" + day)`.
pub fn for_day(day: &str) -> Self {
@ -98,12 +206,84 @@ impl MixParams {
}
}
/// The dataset layout (era layout, `docs/plans/era-layout.md` section 1.2): word `w` of the dataset holds word
/// `j(w)` of item `t(w)`, where `j(w)` gathers the four bits of `w` at the ascending positions `pos` and `t(w)` is
/// `w` with those bits removed. [`Layout::LINEAR`] (`pos = [0, 1, 2, 3]`) is `dataset[w] = item(w >> 4)[w & 15]`,
/// the lottery hash's mapping. Every position is below 16, so the mapping is the same at every dataset size of
/// at least 2^16 words and an item keeps its value at every size.
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
pub struct Layout {
pub pos: [u8; 4],
}
impl Layout {
pub const LINEAR: Layout = Layout { pos: [0, 1, 2, 3] };
pub fn is_linear(&self) -> bool {
self.pos == [0, 1, 2, 3]
}
/// Positions ascending, distinct, below 16.
pub fn is_valid(&self) -> bool {
self.pos.iter().all(|&p| p < 16) && (1..4).all(|i| self.pos[i] > self.pos[i - 1])
}
/// `(t, j)` of word index `w`.
#[inline(always)]
pub fn split(&self, w: u32) -> (u32, u32) {
if self.is_linear() {
return (w >> 4, w & 15);
}
let mut j = 0u32;
for (i, &p) in self.pos.iter().enumerate() {
j |= ((w >> p) & 1) << i;
}
// remove the highest position first so the lower ones stay where they are
let mut t = w;
for &p in self.pos.iter().rev() {
let p = p as u32;
let low = (1u32 << p) - 1;
t = (t & low) | ((t >> (p + 1)) << p);
}
(t, j)
}
/// The word index of word `j` of item `t`: the inverse of [`Layout::split`].
#[inline(always)]
pub fn join(&self, t: u32, j: u32) -> u32 {
if self.is_linear() {
return (t << 4) | (j & 15);
}
// insert the lowest position first: every later position counts the bit just inserted
let mut w = t;
for (i, &p) in self.pos.iter().enumerate() {
let p = p as u32;
let low = (1u32 << p) - 1;
w = ((w >> p) << (p + 1)) | (w & low) | (((j >> i) & 1) << p);
}
w
}
}
impl Default for Layout {
fn default() -> Self {
Layout::LINEAR
}
}
/// Round key `(r + 1) * 0x9E3779B9` mod 2^32.
#[inline(always)]
pub fn round_key(r: usize) -> u32 {
((r + 1) as u32).wrapping_mul(0x9E3779B9)
}
/// The round key of application `j` (0 <= j < m) of round `r` under multiplier `m`: `round_key(r * m + j)`. For
/// `m = 1` this is `round_key(r)`, version 2's key.
#[inline(always)]
pub fn round_key_mult(r: usize, j: usize, m: usize) -> u32 {
round_key(r * m + j)
}
/// `M_r` on 16 words in place: per word `(s ^ (RC + rk)) * MUL`, then one ChaCha-shaped double round with
/// the four column rotations `ROT[0..3]` and the four diagonal rotations `ROT[4..7]`.
#[inline(always)]
@ -122,9 +302,11 @@ pub fn mixer(s: &mut [u32; 16], rk: u32, mp: &MixParams) {
qr(s, 3, 4, 9, 14, r[4], r[5], r[6], r[7]);
}
/// The 256 MiB cache for one day key.
/// The cache for one day key: 2^log2_words words (256 MiB under version 2).
pub struct Cache {
pub key: [u32; 8],
pub log2_words: u32,
line_mask: u32,
words: Vec<u32>,
}
@ -132,6 +314,12 @@ impl Cache {
/// One segment: 64 chained lines written at `cache[seg * 1024 ..]`.
/// `in_j = prev XOR (sigma || K || seg || j || tag)`, `line_j = B(in_j)`, `prev_0 = 0`.
pub fn fill_segment(words: &mut [u32], seg: usize, key: &[u32; 8]) {
Self::fill_segment_tagged(words, seg, key, &CACHE_TAG)
}
/// [`Cache::fill_segment`] with an explicit chain tag: [`CACHE_TAG`] for the cache, [`HOT_TAG`] for the hot
/// table of `docs/plans/hot-table.md` (the same chain, another key and tag).
pub fn fill_segment_tagged(words: &mut [u32], seg: usize, key: &[u32; 8], tag: &[u32; 2]) {
let base = (seg << CACHE_SEGMENT_LOG2_LINES) * 16;
let seg_words = &mut words[base..base + CACHE_LINES_PER_SEGMENT * 16];
let mut prev = [0u32; 16];
@ -141,8 +329,8 @@ impl Cache {
x[4..12].copy_from_slice(key);
x[12] = seg as u32;
x[13] = j as u32;
x[14] = CACHE_TAG[0];
x[15] = CACHE_TAG[1];
x[14] = tag[0];
x[15] = tag[1];
for i in 0..16 {
x[i] ^= prev[i];
}
@ -152,13 +340,22 @@ impl Cache {
}
}
/// The whole cache on the calling thread: 65,536 chains of 64 ChaCha12 blocks, in segment order.
/// The version 2 cache on the calling thread: 65,536 chains of 64 ChaCha12 blocks, in segment order.
pub fn fill(key: [u32; 8]) -> Cache {
let mut words = vec![0u32; CACHE_WORDS];
for seg in 0..CACHE_SEGMENTS {
Self::fill_log2(key, CACHE_LOG2_WORDS as u32)
}
/// A cache of 2^`log2_words` words (26, 27 or 28 under the growth rule; smaller sizes for tests): 2^(log2 - 10)
/// independent chains of 64 lines, the same chain function at every size, so a larger cache's first segments
/// are the smaller cache's segments word for word.
pub fn fill_log2(key: [u32; 8], log2_words: u32) -> Cache {
assert!((10..=30).contains(&log2_words), "cache log2 words must be in 10..=30");
let shape = Shape { mixer_mult: 1, cache_log2_words: log2_words };
let mut words = vec![0u32; shape.cache_words()];
for seg in 0..shape.cache_segments() {
Self::fill_segment(&mut words, seg, &key);
}
Cache { key, words }
Cache { key, log2_words, line_mask: shape.cache_line_mask(), words }
}
pub fn for_day(day: &str) -> Cache {
@ -170,10 +367,27 @@ impl Cache {
&self.words
}
/// Cache line `a` (0 <= a < 2^22) as 16 words.
pub fn lines(&self) -> usize {
self.words.len() >> 4
}
pub fn line_mask(&self) -> u32 {
self.line_mask
}
pub fn segments(&self) -> usize {
self.lines() >> CACHE_SEGMENT_LOG2_LINES
}
/// Cache line `a` (masked to the cache's lines) as 16 words.
#[inline(always)]
pub fn line(&self, a: u32) -> &[u32] {
let o = (a & CACHE_LINE_MASK) as usize * 16;
let o = (a & self.line_mask) as usize * 16;
&self.words[o..o + 16]
}
/// [`Cache::line`] with the mask as a constant (the verifier's hot path, see [`derive_items`]).
#[inline(always)]
pub fn line_const<const LINE_MASK: u32>(&self, a: u32) -> &[u32] {
let o = (a & LINE_MASK) as usize * 16;
&self.words[o..o + 16]
}
@ -183,11 +397,114 @@ impl Cache {
}
}
/// The hot key of an epoch: `seed_words_from_bytes("igneum-hot/" || seed_bytes)`, `seed_bytes` the program seed
/// bytes before any attempt suffix, so every attempt of one epoch shares one table.
pub fn hot_key(seed_bytes: &[u8]) -> [u32; 8] {
let mut b = Vec::with_capacity(HOT_KEY_TAG.len() + seed_bytes.len());
b.extend_from_slice(HOT_KEY_TAG);
b.extend_from_slice(seed_bytes);
crate::seed::seed_words_from_bytes(&b)
}
/// Words of a hot table of `mb` MiB.
pub fn hot_words(mb: u32) -> u32 {
mb * HOT_WORDS_PER_MIB
}
/// Segments of a hot table of `mb` MiB.
pub fn hot_segments(mb: u32) -> u32 {
mb * HOT_SEGMENTS_PER_MIB
}
/// The hot index of a source word: `mulhi(src, words)`, the high 32 bits of the 64-bit product, in `[0, words)`
/// for any table size (the multiply-shift range reduction of spec 01 section 1.13.3).
#[inline(always)]
pub fn hot_index(src: u32, words: u32) -> u32 {
((src as u64 * words as u64) >> 32) as u32
}
/// The hot table `H` of one epoch (hot-table experiment): `mb` MiB of chained ChaCha12 lines under the hot key,
/// read by the hot load slots as `dst ^= H[hot_index(src, words)]`. The verifier holds it beside the cache.
pub struct HotTable {
pub key: [u32; 8],
pub mb: u32,
words: Vec<u32>,
}
impl HotTable {
/// Fill `mb` MiB under `key` on the calling thread.
pub fn fill(key: [u32; 8], mb: u32) -> HotTable {
assert!(mb >= 1 && mb <= 4096, "hot table size in MiB out of range");
let n = hot_words(mb) as usize;
let mut words = vec![0u32; n];
for seg in 0..hot_segments(mb) as usize {
Cache::fill_segment_tagged(&mut words, seg, &key, &HOT_TAG);
}
HotTable { key, mb, words }
}
/// The table of the epoch whose program seed bytes are `seed_bytes`.
pub fn for_seed_bytes(seed_bytes: &[u8], mb: u32) -> HotTable {
Self::fill(hot_key(seed_bytes), mb)
}
#[inline(always)]
pub fn n_words(&self) -> u32 {
self.words.len() as u32
}
/// `H[i]`.
#[inline(always)]
pub fn at(&self, i: u32) -> u32 {
self.words[i as usize]
}
/// `H[hot_index(src, words)]`: what a hot load reads for source word `src`.
#[inline(always)]
pub fn word(&self, src: u32) -> u32 {
self.words[hot_index(src, self.n_words()) as usize]
}
#[inline(always)]
pub fn words(&self) -> &[u32] {
&self.words
}
/// FNV-1a 64 over the table as little-endian bytes (what `vectors.h` carries as `IGNEUM_HOT_FNV64`).
pub fn fnv1a64(&self) -> u64 {
fnv1a64_words(&self.words)
}
}
/// Derive `ts.len()` items into `out`, all chains interleaved round by round so the cache-line misses of
/// independent items overlap in the memory system (`deriveItems` in the Swift).
/// independent items overlap in the memory system (`deriveItems` in the Swift). Under multiplier `m`
/// (`mp.shape.mixer_mult`) round `r` applies `M` with keys `round_key(r m + j)` for `j = 0 .. m - 1` before its
/// one cache read; the final mixer applies `M` with keys `round_key(8 m + j)`. `m = 1` is version 2.
pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) {
// The item loop lives in its own function, one instance per cache size the growth rule can reach with the line
// mask a constant, never inlined into the callers. Inlined into `MemhardCpu::fetch` it ran at 1.33 ms per unit
// against 0.61 out of line (the version 2 verifier, bisected on one core under the measure lock, 5 October 2026,
// `docs/plans/mixer-x4.md` section 6.6: the constant mask alone, or the mask hoisted into a local, or the
// constant with the loop still inlined, all stayed at 1.33; the out-of-line instances read 0.60 to 0.62). Any
// other cache size (tests) takes the instance with the run-time mask.
match cache.log2_words {
26 => derive_items_mask::<{ (1u32 << 22) - 1 }>(ts, mp, cache, out),
27 => derive_items_mask::<{ (1u32 << 23) - 1 }>(ts, mp, cache, out),
28 => derive_items_mask::<{ (1u32 << 24) - 1 }>(ts, mp, cache, out),
29 => derive_items_mask::<{ (1u32 << 25) - 1 }>(ts, mp, cache, out),
30 => derive_items_mask::<{ (1u32 << 26) - 1 }>(ts, mp, cache, out),
_ => derive_items_mask::<0>(ts, mp, cache, out),
}
}
/// [`derive_items`] with the cache line mask as a constant (`LINE_MASK = 0`: the cache's own run-time mask). Kept
/// out of line on purpose (see [`derive_items`]).
#[inline(never)]
fn derive_items_mask<const LINE_MASK: u32>(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) {
let n = ts.len();
debug_assert!(out.len() >= n);
debug_assert!(LINE_MASK == 0 || LINE_MASK == cache.line_mask);
let m = mp.shape.mixer_mult as usize;
for k in 0..n {
let s = &mut out[k];
let t = ts[k];
@ -197,20 +514,24 @@ pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32;
}
}
for r in 0..ITEM_ROUNDS {
let rk = round_key(r);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
for j in 0..m {
let rk = round_key_mult(r, j, m);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
}
}
for s in out[..n].iter_mut() {
let line = cache.line(s[0]);
let line = if LINE_MASK != 0 { cache.line_const::<LINE_MASK>(s[0]) } else { cache.line(s[0]) };
for i in 0..16 {
s[i] ^= line[i];
}
}
}
let rk = round_key(ITEM_ROUNDS);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
for j in 0..m {
let rk = round_key_mult(ITEM_ROUNDS, j, m);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
}
}
}
@ -221,7 +542,7 @@ pub fn derive_item(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] {
out[0]
}
/// The CPU verifier's view of the memory-hard dataset: the mixer parameters and the 256 MiB cache.
/// The CPU verifier's view of the memory-hard dataset: the mixer parameters (with the shape) and the cache.
pub struct MemhardCpu {
pub params: MixParams,
pub cache: Cache,
@ -231,26 +552,41 @@ pub struct MemhardCpu {
pub const FETCH_MAX: usize = 64;
impl MemhardCpu {
/// Version 2 shape.
pub fn new(key: [u32; 8]) -> Self {
Self { params: MixParams::new(key), cache: Cache::fill(key) }
Self::with_shape(key, Shape::V2)
}
pub fn with_shape(key: [u32; 8], shape: Shape) -> Self {
Self { params: MixParams::with_shape(key, shape), cache: Cache::fill_log2(key, shape.cache_log2_words) }
}
pub fn for_day(day: &str) -> Self {
Self::new(day_key(day))
}
/// `dataset[w] = item(w >> 4)[w & 15]`.
pub fn shape(&self) -> Shape {
self.params.shape
}
/// `dataset[w] = item(w >> 4)[w & 15]` (the linear layout).
pub fn word(&self, w: u32) -> u32 {
derive_item(w >> 4, &self.params, &self.cache)[(w & 15) as usize]
self.word_at(Layout::LINEAR, w)
}
/// `dataset[w] = item(t(w))[j(w)]` under `layout` (era layout; the layout is the program's, the cache the
/// day's, so one cache serves every era of a day).
pub fn word_at(&self, layout: Layout, w: u32) -> u32 {
let (t, j) = layout.split(w);
derive_item(t, &self.params, &self.cache)[j as usize]
}
/// `out[k] = dataset[idx[k]]` for every k, `idx.len() <= FETCH_MAX`. Equal items are derived once.
/// Returns the number of distinct items derived.
pub fn fetch(&self, idx: &[u32], out: &mut [u32]) -> usize {
pub fn fetch(&self, idx: &[u32], out: &mut [u32], layout: Layout) -> usize {
let n = idx.len();
assert!(n <= FETCH_MAX && out.len() >= n);
let mut uniq = [0u32; FETCH_MAX];
let mut slot = [0u8; FETCH_MAX];
let mut word = [0u8; FETCH_MAX];
let mut u = 0usize;
for k in 0..n {
let t = idx[k] >> 4;
let (t, j) = layout.split(idx[k]);
word[k] = j as u8;
let found = uniq[..u].iter().position(|&x| x == t);
let j = match found {
Some(j) => j,
@ -265,7 +601,39 @@ impl MemhardCpu {
let mut items = [[0u32; 16]; FETCH_MAX];
derive_items(&uniq[..u], &self.params, &self.cache, &mut items);
for k in 0..n {
out[k] = items[slot[k] as usize][(idx[k] & 15) as usize];
out[k] = items[slot[k] as usize][word[k] as usize];
}
u
}
/// `out[k][j] = dataset[base[k] + j]` for `j < width` (read-width experiment): `base[k]` is aligned to `width`
/// words and the layout's low `log2(width)` positions are the identity, so every lane's words lie in one item
/// at consecutive word offsets, derived once per distinct item. Returns the distinct items.
pub fn fetch_wide(&self, base: &[u32], width: usize, out: &mut [[u32; 16]], layout: Layout) -> usize {
let n = base.len();
assert!(n <= FETCH_MAX && out.len() >= n && width <= 16);
debug_assert!((0..width.trailing_zeros() as usize).all(|i| layout.pos[i] == i as u8), "a wide load needs the identity on its low positions");
let mut uniq = [0u32; FETCH_MAX];
let mut slot = [0u8; FETCH_MAX];
let mut word = [0u8; FETCH_MAX];
let mut u = 0usize;
for k in 0..n {
let (t, j0) = layout.split(base[k]);
word[k] = j0 as u8;
let j = match uniq[..u].iter().position(|&x| x == t) {
Some(j) => j,
None => {
uniq[u] = t;
u += 1;
u - 1
}
};
slot[k] = j as u8;
}
let mut items = [[0u32; 16]; FETCH_MAX];
derive_items(&uniq[..u], &self.params, &self.cache, &mut items);
for k in 0..n {
let o = word[k] as usize;
out[k][..width].copy_from_slice(&items[slot[k] as usize][o..o + width]);
}
u
}
@ -285,6 +653,40 @@ mod tests {
assert_eq!(mp.rc[0], 0xbab68293);
assert_eq!(mp.rc[15], 0x31b49ee2);
assert!(mp.mul.iter().all(|m| m & 1 == 1));
assert_eq!(mp.shape, Shape::V2);
}
/// Era layout: split and join are inverse, the linear layout is today's mapping, and an interleaved layout
/// keeps every position below 16 so the mapping is the same at every size of at least 2^16 words.
#[test]
fn layout_split_join() {
let lin = Layout::LINEAR;
assert!(lin.is_linear() && lin.is_valid());
for w in [0u32, 1, 15, 16, 17, 0x0fff_ffff, 0xffff_ffff] {
assert_eq!(lin.split(w), (w >> 4, w & 15));
assert_eq!(lin.join(w >> 4, w & 15), w);
}
let l = Layout { pos: [0, 1, 7, 12] };
assert!(!l.is_linear() && l.is_valid());
for w in [0u32, 1, 2, 3, 4, 127, 128, 129, 4095, 4096, 0x0fff_ffff, 0x1234_5678, 0xffff_ffff] {
let (t, j) = l.split(w);
assert!(j < 16);
assert_eq!(l.join(t, j), w, "w {w:#x}");
}
// bits: j0 = bit 0, j1 = bit 1, j2 = bit 7, j3 = bit 12; t = the other 28 bits in order
assert_eq!(l.split(0b1_0000_0000_0000), (0, 8));
assert_eq!(l.split(1 << 7), (0, 4));
assert_eq!(l.split(0b100), (1, 0));
// every t in 0..2^(D-4) appears exactly once among w < 2^D (D = 16), with every j
let mut seen = vec![0u32; 1 << 12];
for w in 0..(1u32 << 16) {
let (t, j) = l.split(w);
seen[t as usize] |= 1 << j;
}
assert!(seen.iter().all(|&s| s == 0xffff));
assert!(!Layout { pos: [0, 1, 1, 5] }.is_valid());
assert!(!Layout { pos: [0, 1, 2, 16] }.is_valid());
assert!(!Layout { pos: [1, 0, 2, 3] }.is_valid());
}
#[test]
@ -296,6 +698,40 @@ mod tests {
assert_eq!(y, z);
}
/// Hot-table experiment: the genesis epoch's table (seed bytes "igneum-genesis") as the hot packs carry it
/// (`proto-cuda/packs-ca2-hot/hot32k4/vectors.json`: hot_head, hot_fnv1a64; the head is the same at every size,
/// a larger table is more segments). The index mapping stays inside the table for any size.
#[test]
fn hot_table_fill_vector_and_index() {
assert_ne!(HOT_TAG, CACHE_TAG);
let h = HotTable::for_seed_bytes(b"igneum-genesis", 32);
assert_eq!(h.n_words(), 1 << 23);
assert_eq!(hot_segments(32), 8192);
assert_eq!(
&h.words()[..16],
&[
0x8068cc73, 0x6036ebf9, 0xb604cd25, 0x8ffb840e, 0xc54074a2, 0x285c0695, 0x77512425, 0xc26a58a7,
0x72c88757, 0xc10fca78, 0x513825dd, 0x30d6ccc8, 0x9a05e7cf, 0xb9533f50, 0x4bac3ba0, 0xa5c19528
]
);
assert_eq!(h.fnv1a64(), 0xc1767ba3ef02719f, "hot32k4 pack, hot_fnv1a64");
assert_eq!(h.key, hot_key(b"igneum-genesis"));
assert_ne!(h.key, day_key("2026-10-03"));
// a different seed, a different table; the same seed under the cache tag is not the hot table
assert_ne!(HotTable::for_seed_bytes(b"igneum-genesis\x01\x00\x00\x00", 1).words()[..16], h.words()[..16]);
let mut under_cache_tag = vec![0u32; 1024];
Cache::fill_segment(&mut under_cache_tag, 0, &h.key);
assert_ne!(&under_cache_tag[..16], &h.words()[..16]);
for words in [hot_words(32), hot_words(64), hot_words(96)] {
assert_eq!(hot_index(0, words), 0);
assert!(hot_index(u32::MAX, words) < words);
assert_eq!(hot_index(u32::MAX, words), words - 1);
assert!(hot_index(0x8000_0000, words) == words / 2);
}
assert_eq!(hot_index(0x1234_5678, 1 << 24), 0x1234_5678 >> 8);
assert_eq!(h.word(0x8000_0000), h.at(1 << 22));
}
#[test]
fn first_cache_line_matches_pack() {
// vectors.json cache_head for day 2026-10-03: segment 0, line 0, with prev = 0.
@ -310,4 +746,96 @@ mod tests {
]
);
}
/// Option C: the schedule table of `docs/plans/mixer-x4.md` (day -> doublings, cache words, dataset words at a
/// 2^28 genesis). The doublings fall at years 4 and 12 exactly, never a day early.
#[test]
fn growth_schedule_table() {
let table: [(u64, u32, u32, u32); 12] = [
(0, 0, 26, 28),
(1, 0, 26, 28),
(365, 0, 26, 28),
(1_459, 0, 26, 28),
(1_460, 1, 27, 29),
(2_920, 1, 27, 29),
(4_379, 1, 27, 29),
(4_380, 2, 28, 30),
(10_219, 2, 28, 30),
(10_220, 3, 29, 31),
(21_900, 4, 30, 32),
(100_000, 6, 32, 32),
];
for (d, k, c, s) in table {
assert_eq!(growth_doublings(d), k, "day {d}");
assert_eq!(cache_log2_words(d), c, "day {d}");
assert_eq!(dataset_log2_words(28, d), s, "day {d}");
}
// the designed 2 GiB genesis: 2^29 words, 2^30 at year 4, 2^31 at year 12
assert_eq!(dataset_log2_words(29, 0), 29);
assert_eq!(dataset_log2_words(29, 1_460), 30);
assert_eq!(dataset_log2_words(29, 4_380), 31);
// the linear schedule itself: 2 GiB x (1 + d / 1460) crosses 4 GiB at day 1,460 and 8 GiB at day 4,380
for d in [1_459u64, 1_460, 4_379, 4_380] {
let bytes = 2u64 * (1 << 30) + (1u64 << 29) * d / 365;
let k = (bytes / (2u64 << 30)).ilog2();
assert_eq!(growth_doublings(d), k, "day {d}: linear {bytes} bytes");
}
assert_eq!(days_since_genesis(20_730, 20_729), 1);
assert_eq!(days_since_genesis(20_729, 20_729), 0);
assert_eq!(days_since_genesis(20_000, 20_729), 0);
let v2 = Shape::for_class_day(&LoadClass::V2, 100_000);
assert_eq!(v2, Shape::V2);
let v3 = Shape::for_class_day(&LoadClass::MX4, 0);
assert_eq!(v3, Shape { mixer_mult: 4, cache_log2_words: 26 });
assert_eq!(Shape::for_class_day(&LoadClass::MX4, 1_460).cache_log2_words, 27);
assert_eq!(v3.mixers_per_item(), 36);
assert_eq!(Shape::V2.mixers_per_item(), 9);
assert_eq!(Shape::V2.cache_segments(), CACHE_SEGMENTS);
assert_eq!(Shape::V2.cache_line_mask(), CACHE_LINE_MASK);
assert_eq!(Shape::V2.log2_segments(), 16);
}
/// The multiplied mixer, restated by hand on a small cache: `m` applications with keys `round_key(r m + j)`
/// before every read, the same 8 reads; `m = 1` is `derive_item` of version 2 word for word; a larger cache's
/// first segments equal the smaller cache's.
#[test]
fn mixer_mult_by_hand() {
let key = day_key("2026-10-03");
let small = Cache::fill_log2(key, 16);
let big = Cache::fill_log2(key, 18);
assert_eq!(&big.words()[..small.words().len()], small.words());
assert_eq!(small.segments(), 64);
assert_eq!(small.line_mask(), 4095);
for m in [1u32, 2, 4] {
let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 16 });
for t in [0u32, 1, 12_345, u32::MAX] {
let got = derive_item(t, &mp, &small);
let mut s = [0u32; 16];
s[..8].copy_from_slice(&key);
for i in 0..8 {
s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]);
}
for r in 0..8usize {
for j in 0..m as usize {
mixer(&mut s, round_key(r * m as usize + j), &mp);
}
let line = small.line(s[0]);
for i in 0..16 {
s[i] ^= line[i];
}
}
for j in 0..m as usize {
mixer(&mut s, round_key(8 * m as usize + j), &mp);
}
assert_eq!(got, s, "m {m} t {t}");
}
}
let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16 });
let v3 = MixParams::with_shape(key, Shape { mixer_mult: 4, cache_log2_words: 16 });
assert_ne!(derive_item(0, &v2, &small), derive_item(0, &v3, &small));
assert_eq!(round_key_mult(0, 0, 1), round_key(0));
assert_eq!(round_key_mult(8, 0, 1), round_key(8));
assert_eq!(round_key_mult(2, 3, 4), round_key(11));
assert_eq!(round_key_mult(8, 3, 4), round_key(35));
}
}

397
igneum-pow/src/packcheck.rs Normal file
View file

@ -0,0 +1,397 @@
//! A program pack on disk, read the way the one-click workers read it (`proto-cuda/nvrtc/packfile.h`, `pf_load`),
//! and checked against the seeds the node is on.
//!
//! The rule (5 October 2026, the epoch 34 incident on both PCs): a pack's `IGNEUM_SEEDW_INIT` is the seed words of
//! the program's ATTEMPT, `attempt_words(epoch_seed, IGNEUM_PROGRAM_ATTEMPT)`, not the words of the bare seed. The
//! generator retries a rejected candidate with `seed || k_le32` (spec 01 section 1.4.6), so from attempt 1 on the
//! bare-seed words and the pack's words differ. The workers derived the expected words from the bare seed, refused
//! every pack of a retried program ("the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT") and the miner
//! and the app restarted them forever. Epoch 34 (seed `009858237e11...`) was the first live epoch whose program is
//! a later attempt. This module is the one place that rule is written in Rust; the miner checks every pack it
//! writes with it before a worker sees the pack, and the tests pin the attempt vectors the C side also pins.
use crate::generator::{attempt_words, ProgramClass};
use crate::seed::seed_words_from_bytes;
use std::fmt;
use std::path::Path;
/// What a pack says about itself.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct PackIdentity {
pub epoch_hex: String,
pub day_hex: String,
pub attempt: u32,
pub seedw: [u32; 8],
pub keyw: [u32; 8],
/// `IGNEUM_GENERATOR` (2 or 3; a pack without the line is generator 1, which no worker runs).
pub generator: u32,
/// The program class the generator version names (Counter ASIC 2.0).
pub class: ProgramClass,
/// `IGNEUM_ERA_SEED_HEX` when the pack carries one (class v3 chain packs).
pub era_hex: Option<String>,
}
/// Why a pack is not the one a worker should mine with. `Display` is the plain-words line the logs carry.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum PackFault {
/// program.h or seeds.txt is missing or does not parse.
Unreadable(String),
/// A well-formed pack for other seeds than the node's: the pack is stale (or the node moved on).
OutOfDate { pack_epoch: String, pack_day: String, want_epoch: String, want_day: String },
/// The files of one pack contradict each other (seeds.txt against program.h, or the init words against the
/// seeds and the attempt): a half-written or hand-edited pack, or a worker and an exporter on different rules.
Disagree(String),
/// The pack is of another program class than the one the chain is on (spec 01 section 1.4.5: an implementation
/// refuses a pack whose generator version is not its own), or its era seed is not the era the job names.
WrongClass(String),
}
impl fmt::Display for PackFault {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
match self {
PackFault::Unreadable(w) => write!(f, "program pack unreadable: {w}"),
PackFault::OutOfDate { pack_epoch, pack_day, want_epoch, want_day } => write!(
f,
"program pack out of date: the pack is for epoch {} day {}, the node is on epoch {} day {}",
short(pack_epoch),
day_label(pack_day),
short(want_epoch),
day_label(want_day)
),
PackFault::Disagree(w) => write!(f, "program pack and its seeds disagree: {w}"),
PackFault::WrongClass(w) => write!(f, "program pack of the wrong class: {w}"),
}
}
}
impl std::error::Error for PackFault {}
fn short(hex: &str) -> &str {
if hex.len() >= 16 {
&hex[..16]
} else {
hex
}
}
/// The day bytes are `igneum-day/` followed by the little-endian day index (`bind::day_bytes`); print the index
/// when the hex has that shape, else the hex.
fn day_label(hex: &str) -> String {
const PREFIX: &str = "69676e65756d2d6461792f"; // "igneum-day/"
if let Some(rest) = hex.strip_prefix(PREFIX) {
if let Some(bytes) = unhex(rest) {
let mut v = 0u64;
for (i, b) in bytes.iter().enumerate().take(8) {
v |= (*b as u64) << (8 * i);
}
return v.to_string();
}
}
hex.to_string()
}
pub fn hex(bytes: &[u8]) -> String {
bytes.iter().map(|b| format!("{b:02x}")).collect()
}
fn unhex(s: &str) -> Option<Vec<u8>> {
if s.len() % 2 != 0 {
return None;
}
(0..s.len()).step_by(2).map(|i| u8::from_str_radix(&s[i..i + 2], 16).ok()).collect()
}
/// `#define NAME <rest of line>` in a header; the value as text, trimmed, with a trailing `//` comment removed.
fn define(text: &str, name: &str) -> Option<String> {
for line in text.lines() {
let t = line.trim_start();
let Some(rest) = t.strip_prefix("#define ") else { continue };
let rest = rest.trim_start();
let Some(after) = rest.strip_prefix(name) else { continue };
if !after.starts_with(|c: char| c.is_whitespace()) {
continue;
}
let v = after.trim();
let v = v.split("//").next().unwrap_or("").trim();
return Some(v.to_string());
}
None
}
fn define_str(text: &str, name: &str) -> Option<String> {
let v = define(text, name)?;
let v = v.strip_prefix('"')?.strip_suffix('"')?;
Some(v.to_string())
}
fn define_u32(text: &str, name: &str) -> Option<u32> {
let v = define(text, name)?;
let v = v.trim_end_matches('u');
if let Some(h) = v.strip_prefix("0x") {
u32::from_str_radix(h, 16).ok()
} else {
v.parse().ok()
}
}
fn define_words(text: &str, name: &str) -> Option<[u32; 8]> {
let v = define(text, name)?;
let inner = v.trim().strip_prefix('{')?.strip_suffix('}')?;
let mut out = [0u32; 8];
let mut n = 0;
for part in inner.split(',') {
let p = part.trim().trim_end_matches('u');
if p.is_empty() {
continue;
}
if n >= 8 {
return None;
}
out[n] = if let Some(h) = p.strip_prefix("0x") { u32::from_str_radix(h, 16).ok()? } else { p.parse().ok()? };
n += 1;
}
(n == 8).then_some(out)
}
/// One `key value` line of seeds.txt.
fn seeds_line(text: &str, key: &str) -> Option<String> {
text.lines().find_map(|l| l.strip_prefix(key).and_then(|r| r.strip_prefix(' ')).map(|v| v.trim().to_string()))
}
/// Checks the texts of a pack (program.h, and seeds.txt when it exists) against the seeds a worker will be asked
/// to mine with. Pure: the miner and the tests call it with file contents.
pub fn verify_pack_texts(program_h: &str, seeds_txt: Option<&str>, want_epoch: &[u8], want_day: &[u8]) -> Result<PackIdentity, PackFault> {
verify_pack_texts_chain(program_h, seeds_txt, want_epoch, want_day, None, None)
}
/// [`verify_pack_texts`] that also demands a program class and, for class v3, the era seed the chain is on
/// (Counter ASIC 2.0, 5 October 2026). `want_class` `None` accepts either class; `want_era` `None` skips the era.
/// A pack whose `IGNEUM_GENERATOR` is neither 2 nor 3 is refused whatever is wanted.
pub fn verify_pack_texts_chain(
program_h: &str,
seeds_txt: Option<&str>,
want_epoch: &[u8],
want_day: &[u8],
want_class: Option<ProgramClass>,
want_era: Option<&[u8]>,
) -> Result<PackIdentity, PackFault> {
let generator = define_u32(program_h, "IGNEUM_GENERATOR").unwrap_or(1);
let Some(class) = ProgramClass::from_generator(generator) else {
return Err(PackFault::WrongClass(format!("IGNEUM_GENERATOR {generator} is not a generator version this software runs (2 or 3)")));
};
// IGNEUM_PROGRAM_CLASS, when present, must name the class the generator version names
if let Some(named) = define_str(program_h, "IGNEUM_PROGRAM_CLASS") {
if ProgramClass::parse(&named) != Some(class) {
return Err(PackFault::Disagree(format!("IGNEUM_PROGRAM_CLASS {named:?} does not match IGNEUM_GENERATOR {generator}")));
}
}
let era_hex = define_str(program_h, "IGNEUM_ERA_SEED_HEX").map(|h| h.to_ascii_lowercase());
if let Some(want) = want_class {
if want != class {
return Err(PackFault::WrongClass(format!("the pack is program class {} (generator {generator}), the chain is on class {}", class.name(), want.name())));
}
}
if let (Some(want), ProgramClass::V3) = (want_era, class) {
let want_hex = hex(want);
match &era_hex {
Some(h) if *h == want_hex => {}
Some(h) => return Err(PackFault::WrongClass(format!("the pack's era seed {} is not the era seed {} the job names", short(h), short(&want_hex)))),
None => return Err(PackFault::WrongClass("a class v3 pack without IGNEUM_ERA_SEED_HEX; the job names an era seed".into())),
}
}
let seedw = define_words(program_h, "IGNEUM_SEEDW_INIT").ok_or_else(|| PackFault::Unreadable("program.h has no IGNEUM_SEEDW_INIT with 8 words".into()))?;
let keyw = define_words(program_h, "IGNEUM_KEY_INIT").ok_or_else(|| PackFault::Unreadable("program.h has no IGNEUM_KEY_INIT with 8 words".into()))?;
let attempt = define_u32(program_h, "IGNEUM_PROGRAM_ATTEMPT").unwrap_or(0);
let mut epoch_hex = define_str(program_h, "IGNEUM_SEED_BYTES_HEX").unwrap_or_default();
let mut day_hex = define_str(program_h, "IGNEUM_DAY_BYTES_HEX").unwrap_or_default();
if let Some(s) = seeds_txt {
let e = seeds_line(s, "epoch_seed_hex").ok_or_else(|| PackFault::Unreadable("seeds.txt has no epoch_seed_hex line".into()))?;
let d = seeds_line(s, "day_seed_hex").ok_or_else(|| PackFault::Unreadable("seeds.txt has no day_seed_hex line".into()))?;
if !epoch_hex.is_empty() && !epoch_hex.eq_ignore_ascii_case(&e) {
return Err(PackFault::Disagree(format!("seeds.txt names epoch {} but program.h was generated for epoch {} (a pack half rewritten?)", short(&e), short(&epoch_hex))));
}
if !day_hex.is_empty() && !day_hex.eq_ignore_ascii_case(&d) {
return Err(PackFault::Disagree(format!("seeds.txt names day {} but program.h was generated for day {}", day_label(&d), day_label(&day_hex))));
}
epoch_hex = e.to_ascii_lowercase();
day_hex = d.to_ascii_lowercase();
}
if epoch_hex.is_empty() || day_hex.is_empty() {
return Err(PackFault::Unreadable("no seeds: neither seeds.txt nor IGNEUM_SEED_BYTES_HEX / IGNEUM_DAY_BYTES_HEX in program.h".into()));
}
let epoch_bytes = unhex(&epoch_hex).filter(|b| b.len() == 32).ok_or_else(|| PackFault::Unreadable("the epoch seed is not 32 bytes of hex".into()))?;
let day_bytes = unhex(&day_hex).ok_or_else(|| PackFault::Unreadable("the day seed hex is malformed".into()))?;
// The pack's own consistency first: a pack that contradicts itself is never "out of date", it is broken
let want_w = attempt_words(&epoch_bytes, attempt);
if want_w != seedw {
return Err(PackFault::Disagree(format!(
"IGNEUM_SEEDW_INIT is not attempt {attempt} of the epoch seed {} (the words of attempt {attempt} are {:08x} {:08x} ..., the pack has {:08x} {:08x} ...)",
short(&epoch_hex),
want_w[0],
want_w[1],
seedw[0],
seedw[1]
)));
}
let want_k = seed_words_from_bytes(&day_bytes);
if want_k != keyw {
return Err(PackFault::Disagree(format!("IGNEUM_KEY_INIT is not the key of the day seed {} ", day_label(&day_hex))));
}
let want_epoch_hex = hex(want_epoch);
let want_day_hex = hex(want_day);
if epoch_hex != want_epoch_hex || day_hex != want_day_hex {
return Err(PackFault::OutOfDate { pack_epoch: epoch_hex, pack_day: day_hex, want_epoch: want_epoch_hex, want_day: want_day_hex });
}
Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw, generator, class, era_hex })
}
/// [`verify_pack_texts`] over a pack directory.
pub fn verify_pack_dir(dir: &Path, want_epoch: &[u8], want_day: &[u8]) -> Result<PackIdentity, PackFault> {
verify_pack_dir_chain(dir, want_epoch, want_day, None, None)
}
/// [`verify_pack_texts_chain`] over a pack directory.
pub fn verify_pack_dir_chain(dir: &Path, want_epoch: &[u8], want_day: &[u8], want_class: Option<ProgramClass>, want_era: Option<&[u8]>) -> Result<PackIdentity, PackFault> {
let program_h = std::fs::read_to_string(dir.join("program.h")).map_err(|e| PackFault::Unreadable(format!("cannot read {}/program.h: {e}", dir.display())))?;
let seeds = std::fs::read_to_string(dir.join("seeds.txt")).ok();
verify_pack_texts_chain(&program_h, seeds.as_deref(), want_epoch, want_day, want_class, want_era)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::emit::program_header;
use crate::generator::generate_from_seed_bytes;
use crate::verify::Epoch;
// The two live devnet epochs of 5 October 2026 (epoch 33 mined, epoch 34 refused by the one-click workers)
const EPOCH_33: &str = "bed7ab62cbece66cf791485336d81d90fa1452ffed28ecd8a7416960ef64164c";
const EPOCH_34: &str = "009858237e118f69abc8d096e9b1af21c24539eaecdfd1b896588825660a69ec";
// "igneum-day/" || le64(20731)
const DAY_20731: &str = "69676e65756d2d6461792ffb50000000000000";
fn bytes(h: &str) -> Vec<u8> {
unhex(h).unwrap()
}
/// The attempt vectors the C side pins too (proto-cuda/nvrtc/emu/packfile-test.c): a change to either
/// derivation fails on one side first.
#[test]
fn attempt_words_vectors_shared_with_the_workers() {
let e = bytes(EPOCH_34);
assert_eq!(attempt_words(&e, 0), [0x06af2a61, 0x4d67274e, 0x4ebda738, 0xad1dea73, 0x6233cd8c, 0x50371601, 0x39d0b873, 0x6af024a2]);
assert_eq!(attempt_words(&e, 1), [0x0dcff56b, 0x6b1beb0d, 0x234dc70c, 0xe4016fa9, 0x72397152, 0xb558aa79, 0x3ffb3299, 0x72b9962e]);
// epoch 33's bare words, as the CUDA worker printed them on 5 October ("seed words be5983a6 f750dab7 ...")
assert_eq!(attempt_words(&bytes(EPOCH_33), 0)[..2], [0xbe5983a6, 0xf750dab7]);
}
/// The incident: epoch 34's program is a later attempt, epoch 33's is the bare seed. A worker that derives the
/// words from the bare seed accepts 33 and refuses 34.
#[test]
fn epoch_34_program_is_a_later_attempt() {
let p34 = generate_from_seed_bytes("epoch 34", &bytes(EPOCH_34));
assert!(p34.attempt >= 1, "epoch 34 must be a retried program for the incident to reproduce; attempt {}", p34.attempt);
assert_eq!(p34.seed, attempt_words(&bytes(EPOCH_34), p34.attempt));
assert_ne!(p34.seed, attempt_words(&bytes(EPOCH_34), 0));
let p33 = generate_from_seed_bytes("epoch 33", &bytes(EPOCH_33));
assert_eq!(p33.attempt, 0);
}
fn pack_texts(epoch_hex: &str, day_hex: &str) -> (String, String, u32) {
let (e, d) = (bytes(epoch_hex), bytes(day_hex));
let epoch = Epoch::from_seed_bytes(&e, &d, "test");
let h = program_header(&epoch.program, "test day", &epoch.dataset);
let s = format!("epoch_seed_hex {epoch_hex}\nday_seed_hex {day_hex}\nday_index 20731\n");
(h, s, epoch.program.attempt)
}
/// Known-good: the pack of a retried program verifies against its own seeds, with its attempt.
#[test]
fn known_good_pack_of_a_later_attempt_verifies() {
let (h, s, attempt) = pack_texts(EPOCH_34, DAY_20731);
assert!(attempt >= 1);
let id = verify_pack_texts(&h, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).expect("the pack verifies");
assert_eq!(id.attempt, attempt);
assert_eq!(id.epoch_hex, EPOCH_34);
assert_eq!(id.seedw, attempt_words(&bytes(EPOCH_34), attempt));
// without seeds.txt program.h's own bytes carry the pack
assert!(verify_pack_texts(&h, None, &bytes(EPOCH_34), &bytes(DAY_20731)).is_ok());
}
/// Known-mismatched: a well-formed pack for the previous epoch is "out of date" against the new one, in plain
/// words with both epochs named.
#[test]
fn known_mismatched_pack_is_out_of_date() {
let (h, s, _) = pack_texts(EPOCH_33, DAY_20731);
let err = verify_pack_texts(&h, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err();
assert!(matches!(err, PackFault::OutOfDate { .. }), "{err}");
assert_eq!(err.to_string(), "program pack out of date: the pack is for epoch bed7ab62cbece66c day 20731, the node is on epoch 009858237e118f69 day 20731");
}
/// A pack that contradicts itself is "disagree", never "out of date": seeds.txt of one epoch with program.h of
/// another (a half rewritten directory), or init words that are not the attempt's words (a worker on the old
/// rule would have produced this verdict for every retried program).
#[test]
fn inconsistent_pack_disagrees() {
let (h33, _, _) = pack_texts(EPOCH_33, DAY_20731);
let s34 = format!("epoch_seed_hex {EPOCH_34}\nday_seed_hex {DAY_20731}\n");
let err = verify_pack_texts(&h33, Some(&s34), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err();
assert!(matches!(err, PackFault::Disagree(_)), "{err}");
assert!(err.to_string().starts_with("program pack and its seeds disagree: seeds.txt names epoch 009858237e118f69"), "{err}");
let (h34, s, _) = pack_texts(EPOCH_34, DAY_20731);
let bare = attempt_words(&bytes(EPOCH_34), 0);
let edited = h34.lines().map(|l| if l.starts_with("#define IGNEUM_SEEDW_INIT") { format!("#define IGNEUM_SEEDW_INIT {{ {} }}", bare.iter().map(|w| format!("0x{w:08x}")).collect::<Vec<_>>().join(", ")) } else { l.to_string() }).collect::<Vec<_>>().join("\n");
let err = verify_pack_texts(&edited, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err();
assert!(err.to_string().contains("IGNEUM_SEEDW_INIT is not attempt"), "{err}");
// and a pack with no attempt line at all is read as attempt 0 (the packs before generator version 2)
let no_attempt = h34.lines().filter(|l| !l.starts_with("#define IGNEUM_PROGRAM_ATTEMPT")).collect::<Vec<_>>().join("\n");
assert!(verify_pack_texts(&no_attempt, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).is_err());
}
#[test]
fn day_label_reads_the_index() {
assert_eq!(day_label(DAY_20731), "20731");
assert_eq!(day_label("abcd"), "abcd");
}
/// Counter ASIC 2.0: a class v3 chain pack carries generator 3, the class line and the era seed; it is refused
/// when the chain wants class v2, when the era differs, and a v2 pack is refused when the chain wants v3; a
/// generator this software does not run is refused whatever is wanted.
#[test]
fn program_class_and_era_are_checked() {
let e = bytes(EPOCH_34);
let d = bytes(DAY_20731);
let era = [0x5au8; 32];
let v3 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V3, "class test");
let h3 = program_header(&v3.program, "test day", &v3.dataset);
assert!(h3.contains("#define IGNEUM_GENERATOR 3\n"));
assert!(h3.contains("#define IGNEUM_PROGRAM_CLASS \"v3\"\n"));
assert!(h3.contains(&format!("#define IGNEUM_ERA_SEED_HEX \"{}\"\n", hex(&era))));
let id = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V3), Some(&era)).unwrap();
assert_eq!((id.generator, id.class, id.era_hex.as_deref()), (3, ProgramClass::V3, Some(hex(&era).as_str())));
assert_eq!(id.attempt, v3.program.attempt);
assert!(verify_pack_texts(&h3, None, &e, &d).is_ok(), "no class wanted: either class passes");
let err = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V2), None).unwrap_err();
assert!(matches!(err, PackFault::WrongClass(_)), "{err}");
assert!(err.to_string().contains("program pack of the wrong class"), "{err}");
let err = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V3), Some(&[1u8; 32])).unwrap_err();
assert!(err.to_string().contains("era seed"), "{err}");
// the v2 pack of the same seeds: generator 2, no class line, no era line, refused when v3 is wanted
let v2 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V2, "class test");
let h2 = program_header(&v2.program, "test day", &v2.dataset);
assert!(h2.contains("#define IGNEUM_GENERATOR 2\n"));
assert!(!h2.contains("IGNEUM_PROGRAM_CLASS") && !h2.contains("IGNEUM_ERA_SEED_HEX"));
let plain = Epoch::from_seed_bytes(&e, &d, "class test");
assert_eq!(program_header(&plain.program, "test day", &plain.dataset), h2, "class v2 from the chain is the v2 export byte for byte");
let id = verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V2), Some(&era)).unwrap();
assert_eq!((id.generator, id.class, id.era_hex), (2, ProgramClass::V2, None));
assert!(matches!(verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V3), None), Err(PackFault::WrongClass(_))));
// a generator nobody runs
let h9 = h2.replace("#define IGNEUM_GENERATOR 2\n", "#define IGNEUM_GENERATOR 9\n");
assert!(matches!(verify_pack_texts(&h9, None, &e, &d), Err(PackFault::WrongClass(_))));
// a class line that contradicts the generator
let bad = h3.replace("#define IGNEUM_PROGRAM_CLASS \"v3\"\n", "#define IGNEUM_PROGRAM_CLASS \"v2\"\n");
assert!(matches!(verify_pack_texts(&bad, None, &e, &d), Err(PackFault::Disagree(_))));
}
}

View file

@ -1,10 +1,137 @@
//! The CPU reference interpreter for one 32-lane warp (`cpuWarpTraced` in the Swift) and the API the node
//! calls. Dataset words come from the memory-hard cache (default) or from the closed form (old packs).
use crate::generator::{generate, Instr, Op, Program, ITERATIONS, LANES};
use crate::memhard::MemhardCpu;
use crate::generator::{generate, generate_class, EraParams, Instr, LoadClass, Op, Program, ProgramClass, ITERATIONS, LANES};
use crate::memhard::{hot_index, HotTable, Layout, MemhardCpu, Shape};
use crate::seed::day_key;
/// The load address of an era program (`docs/plans/era-layout.md` section 1.3): `y = rotl(x * M, R)`, then the
/// window of the load site, `k = min(win, D - 26)` (0 when `D <= 26`), `idx = ((y & (MASK >> k)) | ((off &
/// (2^k - 1)) << (D - k))) & MASK`. For every other class `idx = x & MASK`, the lottery hash's address. `mask` is
/// `2^D - 1`. The acceptance mirror calls this at the rule's constant `D = 28`.
#[inline(always)]
pub fn load_index(era: Option<&EraParams>, ins: &Instr, x: u32, mask: u32, log2: u32) -> u32 {
match era {
None => x & mask,
Some(e) => {
let (wm, off) = window(ins, mask, log2);
let y = x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot);
((y & wm) | off) & mask
}
}
}
/// The window of a load site at a dataset of `2^log2` words: `(window mask, offset)` such that
/// `idx = (y & window mask) | offset` lies in the site's aligned window of `2^(log2 - k)` words.
#[inline(always)]
pub fn window(ins: &Instr, mask: u32, log2: u32) -> (u32, u32) {
let k = (ins.win as u32).min(log2.saturating_sub(26));
let wm = mask >> k;
let off = ((ins.off as u32) & ((1u32 << k) - 1)) << (log2 - k);
(wm, off)
}
/// Read-width experiment (5 October 2026): a `load` of `W` words folds every word into `dst`:
/// `x = dst XOR w[0]; for j in 1..W: x = (rotl(x, FOLD_ROT) * FOLD_MUL) XOR w[j]; dst = x`. For `W = 1` this is the
/// lottery hash's `dst XOR dataset[...]`. The fold is state-dependent (the rotate-multiply sits between the words),
/// so no function of the line alone replaces it: two different lines give two different maps of `dst`, and a
/// dataset of folded lines cannot be stored in place of the dataset (see `docs/plans/read-width.md`).
pub const FOLD_ROT: u32 = 11;
pub const FOLD_MUL: u32 = 0x9E3779B1;
/// The fold of `words` into `dst` (at least one word).
#[inline(always)]
pub fn fold_words(dst: u32, words: &[u32]) -> u32 {
let mut x = dst ^ words[0];
for &w in &words[1..] {
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w;
}
x
}
/// Variant 5 (scratch): the fill value of word `j` (0..2) of slot `slot` of lane `lane` of the unit at base nonce
/// `base`, under program seed words `seed`. The scratch of a unit starts as these values; a slot written during
/// the unit's hash holds what was written. Mirrored as `scr_fill` in every emitted kernel.
#[inline(always)]
pub fn scratch_fill(seed: &[u32; 8], base: u32, lane: u32, slot: u32, j: u32) -> u32 {
splitmix32(
(base.wrapping_add(lane) ^ seed[j as usize])
.wrapping_add(slot.wrapping_mul(0x9E3779B1))
.wrapping_add((j + 1).wrapping_mul(0x85EBCA77)),
)
}
/// Variant 5: the 16-byte slot after a read-modify-write that read `w` and folded to `x`: `(x ^ w1, rotl(x, 7) ^ w2,
/// x + w0)` behind the slot's tag.
#[inline(always)]
pub fn scratch_rewrite(x: u32, w: &[u32; 3]) -> [u32; 3] {
[x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])]
}
/// The CPU model of one unit's scratch (variant 5): per lane, the written slots and their words. Unwritten slots
/// read as [`scratch_fill`]. A unit touches at most `scratch ops x 32` slots; a GPU keeps the real scratch per
/// resident warp with a per-unit tag per slot.
pub struct ScratchModel {
slots: usize,
written: Vec<bool>,
data: Vec<[u32; 3]>,
pub reads: usize,
pub writes: usize,
/// Soundness tests (`tests/scratch.rs`, `docs/analysis/scratch-soundness.md`): when `Some`, every
/// read-modify-write is appended as it happened. `None` on every verification path.
pub trace: Option<Vec<ScratchEvent>>,
}
/// One scratch read-modify-write as the interpreter saw it (variant 5 soundness tests).
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub struct ScratchEvent {
pub lane: u8,
pub slot: u32,
/// The slot had been written earlier in this unit (a re-hit): the words read were a rewrite, not the fill.
pub hit: bool,
pub read: [u32; 3],
/// The fold result, the new value of `dst`.
pub x: u32,
pub written: [u32; 3],
}
impl ScratchModel {
pub fn new(slots_per_lane: usize) -> Self {
Self {
slots: slots_per_lane,
written: vec![false; LANES * slots_per_lane],
data: vec![[0; 3]; LANES * slots_per_lane],
reads: 0,
writes: 0,
trace: None,
}
}
/// Read slot `slot` of `lane`, then rewrite it from the fold result `x`. Returns the three words read.
#[inline]
pub fn rmw(&mut self, seed: &[u32; 8], base: u32, lane: usize, slot: u32, dst: u32) -> u32 {
let i = lane * self.slots + slot as usize;
let w = if self.written[i] {
self.data[i]
} else {
[
scratch_fill(seed, base, lane as u32, slot, 0),
scratch_fill(seed, base, lane as u32, slot, 1),
scratch_fill(seed, base, lane as u32, slot, 2),
]
};
let x = fold_words(dst, &w);
let out = scratch_rewrite(x, &w);
if let Some(t) = self.trace.as_mut() {
t.push(ScratchEvent { lane: lane as u8, slot, hit: self.written[i], read: w, x, written: out });
}
self.data[i] = out;
self.written[i] = true;
self.reads += 1;
self.writes += 1;
x
}
}
/// Dataset element, closed form of (day words, index). The original prototype's six-operation element.
#[inline(always)]
pub fn dataset_elem(i: u32, d0: u32, d1: u32) -> u32 {
@ -65,24 +192,55 @@ pub struct DatasetSource {
/// in packs so any implementation can rebuild the key. Empty when the key was given directly.
pub key_bytes: Vec<u8>,
pub dataset: Dataset,
/// The hot table of the epoch (hot-table experiment, `docs/plans/hot-table.md`): `Some` when the program's
/// class has one; filled by [`Epoch::new_class`] and [`Epoch::from_seed_bytes_class`] from the program's seed
/// bytes. A hot load reads `hot[hot_index(src, words)]`.
pub hot: Option<HotTable>,
}
impl DatasetSource {
/// Build the source for a day. Memory-hard mode fills the 256 MiB cache on the calling thread.
pub fn new(day: &str, mode: DatasetMode, log2_words: u32) -> Self {
let mut ds = Self::from_key(day_key(day), mode, log2_words);
Self::new_shape(day, mode, log2_words, Shape::V2)
}
/// [`DatasetSource::new`] with the construction's shape (mixer multiplier, cache size; Counter ASIC 2.0).
pub fn new_shape(day: &str, mode: DatasetMode, log2_words: u32, shape: Shape) -> Self {
let mut ds = Self::from_key_shape(day_key(day), mode, log2_words, shape);
ds.key_bytes = format!("day/{day}").into_bytes();
ds
}
pub fn from_key(key: [u32; 8], mode: DatasetMode, log2_words: u32) -> Self {
Self::from_key_shape(key, mode, log2_words, Shape::V2)
}
/// [`DatasetSource::from_key`] with the construction's shape. Memory-hard mode fills a cache of
/// `2^shape.cache_log2_words` words on the calling thread.
pub fn from_key_shape(key: [u32; 8], mode: DatasetMode, log2_words: u32, shape: Shape) -> Self {
assert!((4..=32).contains(&log2_words), "dataset log2 must be in 4..=32");
let mask = if log2_words == 32 { u32::MAX } else { (1u32 << log2_words) - 1 };
let dataset = match mode {
DatasetMode::ClosedForm => Dataset::ClosedForm { d0: key[0], d1: key[1] },
DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::new(key)),
DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::with_shape(key, shape)),
};
Self { log2_words, mask, key, key_bytes: Vec::new(), dataset }
Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None }
}
/// This source with the hot table of the epoch whose program seed bytes are `seed_bytes` (`mb` MiB).
pub fn with_hot(mut self, seed_bytes: &[u8], mb: u32) -> Self {
self.hot = Some(HotTable::for_seed_bytes(seed_bytes, mb));
self
}
/// The hot table of a program's class, filled from its seed bytes (none for a class without one).
pub fn attach_hot_for(&mut self, program: &Program) {
self.hot = program.class.hot.map(|h| HotTable::for_seed_bytes(&program.seed_bytes, h.mb as u32));
}
/// The shape of the memory-hard construction ([`Shape::V2`] for the closed form, which has none).
pub fn shape(&self) -> Shape {
self.memhard().map(|m| m.shape()).unwrap_or(Shape::V2)
}
pub fn mode(&self) -> DatasetMode {
@ -99,18 +257,23 @@ impl DatasetSource {
}
}
/// `dataset[w & mask]`.
/// `dataset[w & mask]` under the linear layout (the lottery hash).
pub fn word(&self, w: u32) -> u32 {
self.word_at(Layout::LINEAR, w)
}
/// `dataset[w & mask]` under a program's layout (era layout). The closed form has no items and ignores it.
pub fn word_at(&self, layout: Layout, w: u32) -> u32 {
let w = w & self.mask;
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => dataset_elem(w, *d0, *d1),
Dataset::MemoryHard(m) => m.word(w),
Dataset::MemoryHard(m) => m.word_at(layout, w),
}
}
/// `out[k] = dataset[idx[k]]`; indices are already masked. Returns items derived (0 for the closed form).
#[inline]
fn fetch(&self, idx: &[u32; LANES], out: &mut [u32; LANES]) -> usize {
fn fetch(&self, idx: &[u32; LANES], out: &mut [u32; LANES], layout: Layout) -> usize {
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => {
for k in 0..LANES {
@ -118,7 +281,24 @@ impl DatasetSource {
}
0
}
Dataset::MemoryHard(m) => m.fetch(idx, out),
Dataset::MemoryHard(m) => m.fetch(idx, out, layout),
}
}
/// `out[k][j] = dataset[base[k] + j]` for `j < width`; bases are masked and aligned to `width` words
/// (`width` 4 or 16, so a lane's words lie in one item). Returns items derived (0 for the closed form).
#[inline]
fn fetch_wide(&self, base: &[u32; LANES], width: usize, out: &mut [[u32; 16]; LANES], layout: Layout) -> usize {
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => {
for k in 0..LANES {
for j in 0..width {
out[k][j] = dataset_elem(base[k] + j as u32, *d0, *d1);
}
}
0
}
Dataset::MemoryHard(m) => m.fetch_wide(base, width, out, layout),
}
}
}
@ -146,7 +326,23 @@ pub fn interpret_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) ->
/// [`interpret_warp`] with explicit init words `I` (section 1.6 of the spec). The packs use `I = program.seed`;
/// a block uses `I = bind::block_init_words(H, nonce)`.
pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, ds: &DatasetSource) -> WarpResult {
interpret_warp_scratch(program, seed, base_nonce, ds, false).0
}
/// [`interpret_warp_init`] that also returns every scratch read-modify-write of the unit in execution order
/// (lane-minor within an instruction, as the interpreter runs them) when `trace` is set; empty otherwise and for
/// a class without a scratch. For the soundness tests of variant 5 only.
pub fn interpret_warp_scratch(
program: &Program,
seed: &[u32; 8],
base_nonce: u32,
ds: &DatasetSource,
trace: bool,
) -> (WarpResult, Vec<ScratchEvent>) {
let mask = ds.mask;
let log2 = ds.log2_words;
let era = program.class.era;
let layout = program.class.layout();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
let nonce = base_nonce.wrapping_add(lane as u32);
@ -160,10 +356,29 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32,
let mut items_derived = 0usize;
let mut idx = [0u32; LANES];
let mut val = [0u32; LANES];
let mut scratch = if program.has_scratch() { Some(ScratchModel::new(program.class.scratch_slots_per_lane())) } else { None };
if trace {
if let Some(m) = scratch.as_mut() {
m.trace = Some(Vec::new());
}
}
let slot_mask = program.class.scratch_slot_mask();
if program.has_hot() {
let h = ds.hot.as_ref().expect("a hot-table program needs the epoch's hot table on the dataset source");
assert_eq!(h.n_words(), program.hot_words(), "the hot table's size is the class's");
}
for _ in 0..ITERATIONS {
let sel = r[0];
for ins in &program.instrs {
step(ins, &mut r, &sel, mask, ds, &mut idx, &mut val, &mut items_derived);
step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived);
if ins.op == Op::Scratch {
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
let (d, a) = (ins.dst as usize, ins.src as usize);
for lane in 0..LANES {
let slot = r[a][lane] & slot_mask;
r[d][lane] = m.rmw(&program.seed, base_nonce, lane, slot, r[d][lane]);
}
}
}
}
let mut hashes = [0u64; LANES];
@ -172,7 +387,8 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32,
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
hashes[lane] = ((hi as u64) << 32) | lo as u64;
}
WarpResult { hashes, items_derived }
let events = scratch.and_then(|m| m.trace).unwrap_or_default();
(WarpResult { hashes, items_derived }, events)
}
#[inline(always)]
@ -182,6 +398,9 @@ fn step(
r: &mut [[u32; LANES]; 8],
sel: &[u32; LANES],
mask: u32,
log2: u32,
era: Option<&EraParams>,
layout: Layout,
ds: &DatasetSource,
idx: &mut [u32; LANES],
val: &mut [u32; LANES],
@ -255,22 +474,46 @@ fn step(
r[d][lane] ^= src[lane ^ m];
}
}
Op::Load => {
Op::Load if ins.width == 1 => {
for lane in 0..LANES {
idx[lane] = r[a][lane] & mask;
idx[lane] = load_index(era, ins, r[a][lane], mask, log2);
}
*items_derived += ds.fetch(idx, val);
*items_derived += ds.fetch(idx, val, layout);
for lane in 0..LANES {
r[d][lane] ^= val[lane];
}
}
Op::Load => {
// Read-width experiment: `width` words from the aligned address, every word folded into dst.
let width = ins.width as usize;
let align = !(ins.width as u32 - 1);
for lane in 0..LANES {
idx[lane] = load_index(era, ins, r[a][lane], mask, log2) & align;
}
let mut vals = [[0u32; 16]; LANES];
*items_derived += ds.fetch_wide(idx, width, &mut vals, layout);
for lane in 0..LANES {
r[d][lane] = fold_words(r[d][lane], &vals[lane][..width]);
}
}
Op::Scratch => {
// handled by the caller (interpret_warp_init), which owns the unit's scratch model
}
Op::Hot => {
// Hot-table experiment: one word of the epoch table at the multiply-shift index, plain xor fold.
let h = ds.hot.as_ref().expect("a hot load needs the hot table");
let n = h.n_words();
for lane in 0..LANES {
r[d][lane] ^= h.at(hot_index(r[a][lane], n));
}
}
Op::WLoad => {
// Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l.
let base = (r[a][0] & mask) & !31;
for lane in 0..LANES {
idx[lane] = base + lane as u32;
}
*items_derived += ds.fetch(idx, val);
*items_derived += ds.fetch(idx, val, Layout::LINEAR);
for lane in 0..LANES {
r[d][lane] ^= val[lane];
}
@ -294,11 +537,39 @@ pub struct Epoch {
/// Default dataset size: 2^28 words = 1 GiB.
pub const DEFAULT_DATASET_LOG2: u32 = 28;
/// Days a day index lies after the network's genesis day (0 for the genesis day and any day before it). The node's
/// entry; the same function as `memhard::days_since_genesis`.
pub fn days_since_genesis(day_index: u64, genesis_day_index: u64) -> u64 {
crate::memhard::days_since_genesis(day_index, genesis_day_index)
}
impl Epoch {
pub fn new(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32) -> Self {
Self { program: generate(seed), dataset: DatasetSource::new(day, mode, dataset_log2) }
}
/// [`Epoch::new`] with a load class (read-width experiment; Counter ASIC 2.0: the class's mixer multiplier
/// shapes the dataset, the cache is the genesis size since a string day has no day index).
pub fn new_class(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32, class: LoadClass) -> Self {
Self::new_class_day(seed, day, mode, dataset_log2, class, 0)
}
/// [`Epoch::new_class`] on day `days_since_genesis` of the growth schedule (the cache of
/// `memhard::cache_log2_words` for a class with the growth rule; the dataset size is the caller's).
pub fn new_class_day(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32, class: LoadClass, days_since_genesis: u64) -> Self {
let shape = Shape::for_class_day(&class, days_since_genesis);
let program = generate_class(seed, class);
let mut dataset = DatasetSource::new_shape(day, mode, dataset_log2, shape);
// hot-table experiment: a hot class fills its table from the seed bytes
dataset.attach_hot_for(&program);
Self { program, dataset }
}
/// `dataset[w]` as this epoch's program reads it: under the program's layout (era layout; linear for v2).
pub fn dataset_word(&self, w: u32) -> u32 {
self.dataset.word_at(self.program.class.layout(), w)
}
/// The production shape: memory-hard, 1 GiB dataset.
pub fn memory_hard(seed: &str, day: &str) -> Self {
Self::new(seed, day, DatasetMode::MemoryHard, DEFAULT_DATASET_LOG2)
@ -309,13 +580,71 @@ impl Epoch {
/// `seed_words_from_bytes(day_bytes)` (`bind::day_bytes`). Memory-hard, 1 GiB dataset. `label` is only
/// recorded in emitted packs.
pub fn from_seed_bytes(epoch_seed: &[u8], day_bytes: &[u8], label: &str) -> Self {
let program = crate::generator::generate_from_seed_bytes(label, epoch_seed);
Self::from_seed_bytes_class(epoch_seed, day_bytes, label, LoadClass::V2)
}
/// [`Epoch::from_seed_bytes`] with a load class (read-width experiment; Counter ASIC 2.0: the class's mixer
/// multiplier shapes the dataset). Day 0 of the growth schedule: the 2^26-word cache and the 2^28-word dataset,
/// which is every devnet pack and vector. A node past the first doubling calls [`Epoch::from_seed_bytes_day`].
pub fn from_seed_bytes_class(epoch_seed: &[u8], day_bytes: &[u8], label: &str, class: LoadClass) -> Self {
Self::from_seed_bytes_day(epoch_seed, day_bytes, label, class, 0, DEFAULT_DATASET_LOG2)
}
/// The chain's shape on day `days_since_genesis` (`memhard::days_since_genesis(day_index(header), day_index(genesis))`,
/// the node's two day indices): the program of the class, and under the class's growth rule the cache of
/// `memhard::cache_log2_words(d)` and the dataset of `memhard::dataset_log2_words(genesis_dataset_log2, d)`
/// (the genesis size is 28 for the 1 GiB devnet, 29 for the designed 2 GiB). Without the growth rule the cache
/// is 2^26 words and the dataset `2^genesis_dataset_log2` on every day.
pub fn from_seed_bytes_day(epoch_seed: &[u8], day_bytes: &[u8], label: &str, class: LoadClass, days_since_genesis: u64, genesis_dataset_log2: u32) -> Self {
let program = crate::generator::generate_from_seed_bytes_class(label, epoch_seed, class);
let key = crate::seed::seed_words_from_bytes(day_bytes);
let mut dataset = DatasetSource::from_key(key, DatasetMode::MemoryHard, DEFAULT_DATASET_LOG2);
let shape = Shape::for_class_day(&class, days_since_genesis);
let dataset_log2 = if class.growth { crate::memhard::dataset_log2_words(genesis_dataset_log2, days_since_genesis) } else { genesis_dataset_log2 };
let mut dataset = DatasetSource::from_key_shape(key, DatasetMode::MemoryHard, dataset_log2, shape);
dataset.key_bytes = day_bytes.to_vec();
dataset.attach_hot_for(&program);
Self { program, dataset }
}
/// The chain's shape with the program class (Counter ASIC 2.0, 5 October 2026): what the node's engine and the
/// miner's pack export build from the seeds a block template carries. Class v2 is [`Epoch::from_seed_bytes`]
/// exactly (the era bytes are ignored and not recorded); class v3 draws from [`crate::generator::V3_CLASS`]
/// with generator version 3 and records the era seed bytes (`E_n`) in the program for the pack.
pub fn from_chain_seeds(epoch_seed: &[u8], day_bytes: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, label: &str) -> Self {
Self { program: Self::chain_program(epoch_seed, era_bytes, class, label), dataset: Self::chain_dataset(day_bytes, class) }
}
/// The program alone of [`Epoch::from_chain_seeds`] (no cache fill): for an engine that shares the day's cache.
pub fn chain_program(epoch_seed: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, label: &str) -> Program {
crate::generator::generate_from_seed_bytes_program_class(label, epoch_seed, class, era_bytes)
}
/// The day's cache and dataset of [`Epoch::from_chain_seeds`], the one entry the node's engine builds a day
/// cache through. The class is an argument because the Counter ASIC 2.0 integration gives class v3 its own item
/// construction (the mixer multiplier) and cache size schedule (ca2-mixer); today both classes build the day of
/// [`Epoch::from_seed_bytes`], and the engine keys its day caches on `(day, class)` so the two never share one.
pub fn chain_dataset(day_bytes: &[u8], class: ProgramClass) -> DatasetSource {
Self::chain_dataset_day(day_bytes, class, 0, DEFAULT_DATASET_LOG2)
}
/// [`Epoch::chain_dataset`] with the day's position since genesis and the network's genesis dataset size: the
/// entry the node's engine and the miner's export build every day cache through, so the cache growth schedule
/// of spec 01 section 1.13.3 has one place to act (ca2-mixer, 5 October 2026, `docs/plans/mixer-x4.md`): the
/// class's load class gives the mixer multiplier and whether the growth rule applies (`Shape::for_class_day`);
/// under the rule the cache is `2^memhard::cache_log2_words(d)` words and the dataset
/// `2^memhard::dataset_log2_words(genesis_dataset_log2, d)`; without it (class v2) the cache is 2^26 words and
/// the dataset the genesis size on every day. `days_since_genesis` is [`days_since_genesis`] of the block's and
/// the genesis header's day indices.
pub fn chain_dataset_day(day_bytes: &[u8], class: ProgramClass, days_since_genesis: u64, genesis_dataset_log2: u32) -> DatasetSource {
let lc = class.load_class();
let shape = Shape::for_class_day(&lc, days_since_genesis);
let dataset_log2 = if lc.growth { crate::memhard::dataset_log2_words(genesis_dataset_log2, days_since_genesis) } else { genesis_dataset_log2 };
let key = crate::seed::seed_words_from_bytes(day_bytes);
let mut dataset = DatasetSource::from_key_shape(key, DatasetMode::MemoryHard, dataset_log2, shape);
dataset.key_bytes = day_bytes.to_vec();
dataset
}
/// The 32 hashes of the warp starting at `base_nonce`.
pub fn hash_warp(&self, base_nonce: u32) -> [u64; LANES] {
hash_warp(&self.program, base_nonce, &self.dataset)
@ -356,6 +685,190 @@ mod tests {
assert_eq!(ds.word(0x0fffffff), 0xf78c84a4);
}
/// Read-width experiment: the fold for one word is a plain xor; a wide fetch hands each lane the words the
/// scalar path would; two distinct lines give two distinct maps of dst (one point suffices as a smoke check).
#[test]
fn fold_and_wide_fetch() {
assert_eq!(fold_words(0x1234_5678, &[0xdead_beef]), 0x1234_5678 ^ 0xdead_beef);
let w = [1u32, 2, 3, 4];
let x = fold_words(7, &w);
let mut y: u32 = 7 ^ 1;
for &v in &w[1..] {
y = y.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ v;
}
assert_eq!(x, y);
assert_ne!(fold_words(7, &[1, 2, 3, 4]), fold_words(7, &[1, 2, 3, 5]));
let ds = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 20);
let mut base = [0u32; LANES];
for (k, b) in base.iter_mut().enumerate() {
*b = ((k as u32).wrapping_mul(0x9E37_79B1) & ds.mask) & !15;
}
let mut out = [[0u32; 16]; LANES];
let items = ds.fetch_wide(&base, 16, &mut out, Layout::LINEAR);
assert!(items >= 1 && items <= LANES);
for k in 0..LANES {
for j in 0..16 {
assert_eq!(out[k][j], ds.word(base[k] + j as u32), "lane {k} word {j}");
}
}
let mut base4 = base;
for b in base4.iter_mut() {
*b += 8;
}
let items4 = ds.fetch_wide(&base4, 4, &mut out, Layout::LINEAR);
assert_eq!(items4, items);
for k in 0..LANES {
for j in 0..4 {
assert_eq!(out[k][j], ds.word(base4[k] + j as u32));
}
}
}
/// A wide-load program interprets identically on the closed form and through the memory-hard path's fold
/// (the same fold code), and a mixed-class epoch builds and hashes.
/// Variant 5: a fill word is deterministic, a rewrite changes the slot, and a second read of a written slot
/// returns the rewrite, not the fill.
#[test]
fn scratch_model() {
let seed = [1u32, 2, 3, 4, 5, 6, 7, 8];
assert_eq!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 1));
assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2));
assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 64, 3, 100, 1));
let mut m = ScratchModel::new(256);
let w = [scratch_fill(&seed, 32, 3, 100, 0), scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2)];
let x = m.rmw(&seed, 32, 3, 100, 0xabcd);
assert_eq!(x, fold_words(0xabcd, &w));
let x2 = m.rmw(&seed, 32, 3, 100, 0xabcd);
assert_eq!(x2, fold_words(0xabcd, &scratch_rewrite(x, &w)));
assert_eq!(m.reads, 2);
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::scratch(4, 128));
assert_eq!(e.program.scratch_ops_per_hash(), 32);
assert_eq!(e.hash_warp(0), e.hash_warp(0));
}
/// Era layout: the load address stays inside the site's window and below the mask at every dataset size (the
/// window floor of 2^26 words clamps the shrink), the interleaved memory-hard dataset reads item(t(w))[j(w)] and
/// is the same prefix at 2^20 and 2^22 words, the wide fetch agrees word for word, and an era epoch hashes
/// deterministically through the interpreter and the single-nonce API.
#[test]
fn era_windows_layout_and_epochs() {
let eb = EraParams::test_era_bytes("igneum-era-test/1");
let c = LoadClass::era(LoadClass::V2, &eb, &[1]);
let e = c.era.unwrap();
let mut ins = Instr { op: Op::Load, dst: 0, src: 1, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 2, off: 3 };
let mut s = crate::seed::SplitMix64::new(7);
for log2 in [20u32, 26, 27, 28, 29] {
let mask = (1u64 << log2) as u32 - 1;
let k = ins.win.min(log2.saturating_sub(26) as u8) as u32;
for _ in 0..1000 {
let x = s.next() as u32;
let idx = load_index(Some(&e), &ins, x, mask, log2);
assert!(idx <= mask);
let (wm, off) = window(&ins, mask, log2);
assert_eq!(idx & !wm, off, "log2 {log2}");
assert_eq!(wm, mask >> k);
assert_eq!(idx, ((x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot) & wm) | off) & mask);
}
}
ins.win = 0;
assert_eq!(load_index(None, &ins, 0xdead_beef, 0x0fff_ffff, 28), 0xdead_beef & 0x0fff_ffff);
// the interleaved dataset: one day cache, the layout per program
let l = e.layout();
assert_eq!(l.pos, [1, 3, 8, 13]);
let small = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 20);
let big = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 22);
let m = small.memhard().unwrap();
for w in [0u32, 1, 4, 5, 255, 256, 4095, 8192, 0x0f_ffff] {
let (t, j) = l.split(w);
assert_eq!(small.word_at(l, w), crate::memhard::derive_item(t, &m.params, &m.cache)[j as usize], "w {w}");
assert_eq!(small.word_at(l, w), big.word_at(l, w), "prefix at w {w}");
assert_eq!(small.word(w), small.word_at(Layout::LINEAR, w));
}
let mut idx = [0u32; LANES];
for (k, i) in idx.iter_mut().enumerate() {
*i = (k as u32).wrapping_mul(0x9E37_79B1) & small.mask;
}
let mut out = [0u32; LANES];
small.fetch(&idx, &mut out, l);
for k in 0..LANES {
assert_eq!(out[k], small.word_at(l, idx[k]));
}
// a 16-byte era: the wide fetch keeps a lane's four words in one item
let eb3 = EraParams::test_era_bytes("igneum-era-test/3");
let c3 = LoadClass::era(LoadClass::fixed(4, 16), &eb3, &[4]);
let l3 = c3.layout();
assert_eq!(l3.pos[..2], [0, 1]);
let mut base = [0u32; LANES];
for (k, b) in base.iter_mut().enumerate() {
*b = ((k as u32).wrapping_mul(0x9E37_79B1) & small.mask) & !3;
}
let mut wide = [[0u32; 16]; LANES];
small.fetch_wide(&base, 4, &mut wide, l3);
for k in 0..LANES {
for j in 0..4 {
assert_eq!(wide[k][j], small.word_at(l3, base[k] + j as u32), "lane {k} word {j}");
}
}
// era epochs hash deterministically, differ per era, and the single-nonce API agrees with the warp
let mut seen = std::collections::HashSet::new();
for (n, c) in [(1u64, c), (3, c3)] {
let ep = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::MemoryHard, 20, c);
assert_eq!(ep.dataset_word(5), ep.dataset.word_at(c.layout(), 5));
let a = ep.hash_warp(64);
assert_eq!(a, ep.hash_warp(64));
assert_eq!(ep.hash(64 + 5), a[5]);
assert!(seen.insert(a[0]), "era {n}");
let closed = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, c);
assert_ne!(closed.hash_warp(64), a);
}
}
/// Hot-table experiment: an epoch of a hot class carries the table, hashes deterministically and differs from
/// version 2; the reference interpreter agrees with a hand-stepped hot load; a hot program without its table is
/// refused.
#[test]
fn hot_epochs_hash() {
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(32, 4));
let h = e.dataset.hot.as_ref().expect("the epoch fills the hot table");
assert_eq!(h.n_words(), 1 << 23);
assert_eq!(h.key, crate::memhard::hot_key(b"igneum-genesis"));
let v2 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::V2);
let a = e.hash_warp(0);
assert_eq!(a, e.hash_warp(0));
assert_ne!(a, v2.hash_warp(0));
assert_ne!(a[0], a[1]);
// the same program under a 64 MiB table reads other words
let e64 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(64, 4));
assert_eq!(e64.program.instrs, e.program.instrs);
assert_ne!(e64.hash_warp(0), a);
// from seed bytes, the chain's shape, with a hot class
let genesis = crate::bind::unhex("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07").unwrap();
let ec = Epoch::from_seed_bytes_class(&genesis, &crate::bind::day_bytes(20_730), "devnet", LoadClass::hot(32, 2));
assert_eq!(ec.dataset.hot.as_ref().unwrap().key, crate::memhard::hot_key(&genesis));
assert_eq!(ec.hash_warp(0), ec.hash_warp(0));
}
#[test]
#[should_panic(expected = "needs the epoch's hot table")]
fn hot_program_without_a_table_is_refused() {
let p = generate_class("igneum-genesis", LoadClass::hot(32, 4));
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 20);
let _ = hash_warp(&p, 0, &ds);
}
#[test]
fn wide_class_epochs_hash() {
for name in ["w16", "w64x4", "50,35,15"] {
let c = LoadClass::parse(name).unwrap();
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, c);
assert_eq!(e.program.class, c);
let a = e.hash_warp(0);
let b = e.hash_warp(0);
assert_eq!(a, b);
assert_ne!(a[0], a[1]);
}
}
#[test]
fn closed_form_genesis_vector_lane0() {
// Generator v2 vectors (4 October 2026), proto-cuda/packs/igneum-genesis/vectors.json.

247
igneum-pow/tests/mixer.rs Normal file
View file

@ -0,0 +1,247 @@
//! The class v3 dataset construction (mixer x4, cache growth option C; `docs/plans/mixer-x4.md`): the soundness
//! runs the brief asks for, on the CPU, with the packs for the GPU runs written on request.
//!
//! 1. Fuzz: `IGNEUM_MIXER_FUZZ` (default 200) programs through the seam (`ProgramClass::V3`), the contract on every
//! instruction (the v2 program of the seed, instruction for instruction), 4 units each across the 32-bit range
//! including the wrap, interpreted twice on the CPU; with `IGNEUM_MIXER_PACKS_OUT=<dir>` every program is written
//! as a pack with its 4 bases in vectors.json for `packbench` and the OpenCL host (the Metal fuzz).
//! 2. Stats: bit balance and single-bit-flip avalanche of the v3 hash against v2 on the same programs and nonces.
//! 3. Edge: the dataset at word 0, word MASK and the item boundary, derived through the interpreter's fetch path and
//! by hand at every multiplier 1, 2, 4, 8, on a small cache.
//! 4. Determinism: two independent epochs of the same seed and day agree on every vector and every emitted file.
use igneum_pow::emit::{export_pack, vectors_json};
use igneum_pow::generator::{generate_from_seed_bytes, generate_from_seed_bytes_class, generate_from_seed_bytes_program_class, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION_V3, INSTR_COUNT, V3_CLASS};
use igneum_pow::memhard::{derive_item, mixer, round_key, Cache, MixParams, Shape};
use igneum_pow::seed::{day_key, SplitMix64};
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch};
use std::collections::HashMap;
use std::path::PathBuf;
const DAY: &str = "2026-10-03";
fn contract(p: &Program, seed: &str, class: LoadClass) {
if class == V3_CLASS {
assert_eq!(p.generator, GENERATOR_VERSION_V3);
}
assert_eq!(p.class, class);
assert_eq!(p.instrs.len(), INSTR_COUNT);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16);
for (k, i) in p.instrs.iter().enumerate() {
assert!(i.src != i.dst, "#{k}: src == dst");
assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot);
assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask);
assert!(i.dst < 8 && i.src < 8 && i.src2 < 8);
assert_eq!(i.width, 1, "#{k}: a v3 load reads one word");
}
assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program");
let v2 = generate_from_seed_bytes(seed, seed.as_bytes());
assert_eq!(p.instrs, v2.instrs, "the v2 program of the seed under the v3 construction");
assert_eq!(p.attempt, v2.attempt);
}
/// Write a pack whose vectors.json carries `bases` instead of the three standard bases (packbench and the OpenCL
/// host check every unit standalone and the ones inside the batch window).
fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) {
let mut pack = export_pack(e, day, source);
let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect();
let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true);
for f in pack.files.iter_mut() {
if f.0 == "vectors.json" {
f.1 = vj.clone();
}
}
pack.write_to(dir).unwrap();
}
#[test]
fn fuzz_v3_programs_cpu() {
let n: usize = std::env::var("IGNEUM_MIXER_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200);
let out = std::env::var("IGNEUM_MIXER_PACKS_OUT").ok().map(PathBuf::from);
// IGNEUM_MIXER_CLASS=mx8 fuzzes the x8 candidate as a load class (generator 2 with the class in the id); the
// default is V3_CLASS through the seam
let class = std::env::var("IGNEUM_MIXER_CLASS").ok().map(|s| LoadClass::parse(&s).expect("a load class")).unwrap_or(V3_CLASS);
let mut rng = SplitMix64::new(0x6967_6e65_756d_2d6d); // "igneum-m"
let shape = Shape::for_class(&class);
assert_eq!(shape.cache_log2_words, 26);
assert!(shape.mixer_mult > 1);
// one memory-hard source per dataset size (the 256 MiB cache fill is 0.2 s each)
let mut mh: HashMap<u32, DatasetSource> = HashMap::new();
let mut manifest = String::from("pack\tlog2\tprogram_id\tbases\n");
let mut units = 0usize;
let mut wraps = 0usize;
for i in 0..n {
let seed = format!("igneum-mixer-fuzz/{i}");
let p = if class == V3_CLASS {
generate_from_seed_bytes_program_class(&seed, seed.as_bytes(), ProgramClass::V3, None)
} else {
generate_from_seed_bytes_class(&seed, seed.as_bytes(), class)
};
contract(&p, &seed, class);
let b0 = (rng.below(8) as u32) * 32;
let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32);
let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32);
let b3 = (rng.next() as u32) & !31;
let bases = [b0, b1, b2, b3];
wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count();
let log2 = [24u32, 26, 28][rng.below(3) as usize];
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, log2, shape));
let e = Epoch { program: p, dataset: ds };
for &b in &bases {
let r1 = e.interpret_warp(b);
let r2 = e.interpret_warp(b);
assert_eq!(r1.hashes, r2.hashes);
assert!(r1.items_derived >= 120 * 32 / 32 && r1.items_derived <= 4_096, "{seed}: {} items", r1.items_derived);
units += 1;
}
if let Some(dir) = &out {
let pack_name = format!("fuzz-{i:03}-{}-l{log2}", class.name());
write_pack_with_bases(&dir.join(&pack_name), &e, DAY, &bases, "igneum-pow tests/mixer.rs fuzz");
manifest.push_str(&format!(
"{pack_name}\t{log2}\t{:016x}\t{}\n",
e.program.program_id(),
bases.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
));
}
mh.insert(log2, e.dataset);
}
println!("fuzz: {n} {} programs, {units} units on the CPU, {wraps} units in the top 256 nonces", class.name());
assert_eq!(units, 4 * n);
assert_eq!(wraps, n);
if let Some(dir) = &out {
std::fs::create_dir_all(dir).unwrap();
std::fs::write(dir.join("manifest.tsv"), manifest).unwrap();
println!("packs written to {}", dir.display());
}
}
/// Bit balance and avalanche of the v3 hash beside v2 on the same program (the TESTS.md section 3 shape, on the
/// CPU, 2^13 nonces per seed): every output bit within 5 sigma of half ones; a single nonce-bit flip moves 50 percent
/// of the output bits within 2 points; no duplicate among the outputs.
#[test]
fn stats_v3_against_v2() {
let n_warps = 256usize; // 8,192 nonces
for seed in ["igneum-genesis", "igneum-genesis/stats1"] {
let v3 = Epoch {
program: generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V3, None),
dataset: DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 24, Shape::for_class(&V3_CLASS)),
};
let v2 = Epoch::new(seed, DAY, DatasetMode::MemoryHard, 24);
for (name, e) in [("v3", &v3), ("v2", &v2)] {
let mut ones = [0u64; 64];
let mut outs = Vec::with_capacity(n_warps * 32);
for w in 0..n_warps {
let h = e.hash_warp(w as u32 * 32);
for &x in &h {
outs.push(x);
for b in 0..64 {
ones[b] += (x >> b) & 1;
}
}
}
let total = (n_warps * 32) as f64;
let sigma = (total / 4.0).sqrt();
for (b, &c) in ones.iter().enumerate() {
let z = (c as f64 - total / 2.0).abs() / sigma;
assert!(z < 5.0, "{seed} {name}: bit {b} ones {c} of {total}, z {z:.2}");
}
// avalanche: flip one bit of the nonce within the unit (lanes 0..31 differ in the low 5 bits) and across
// units (bit 5 and up): compare lane l of unit u with lane l ^ (1 << k) and with unit u ^ (1 << k)
let mut flips = 0u64;
let mut moved = 0u64;
for w in 0..64usize {
let h = e.hash_warp(w as u32 * 32);
for k in 0..5 {
for l in 0..32usize {
moved += (h[l] ^ h[l ^ (1 << k)]).count_ones() as u64;
flips += 1;
}
}
let h2 = e.hash_warp((w ^ 1) as u32 * 32);
for l in 0..32usize {
moved += (h[l] ^ h2[l]).count_ones() as u64;
flips += 1;
}
}
let avg = moved as f64 / flips as f64 / 64.0 * 100.0;
assert!((avg - 50.0).abs() < 2.0, "{seed} {name}: avalanche {avg:.2} percent");
outs.sort_unstable();
let dups = outs.windows(2).filter(|p| p[0] == p[1]).count();
assert_eq!(dups, 0, "{seed} {name}: duplicate outputs");
println!("{seed} {name}: {} outputs, avalanche {avg:.2} percent, worst bit z {:.2}", outs.len(), ones.iter().map(|&c| (c as f64 - total / 2.0).abs() / sigma).fold(0.0, f64::max));
}
assert_ne!(v3.hash_warp(0), v2.hash_warp(0));
}
}
/// The dataset edges under every multiplier on a small cache: word 0, word MASK, the last word of item 0 and the
/// first of item 1, through `DatasetSource::word` and by hand.
#[test]
fn edge_items_every_multiplier() {
let key = day_key(DAY);
let cache = Cache::fill_log2(key, 14);
for m in [1u32, 2, 4, 8] {
let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 14 });
let by_hand = |t: u32| -> [u32; 16] {
let mut s = [0u32; 16];
s[..8].copy_from_slice(&key);
for i in 0..8 {
s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]);
}
for r in 0..8usize {
for j in 0..m as usize {
mixer(&mut s, round_key(r * m as usize + j), &mp);
}
let line = cache.line(s[0]);
for i in 0..16 {
s[i] ^= line[i];
}
}
for j in 0..m as usize {
mixer(&mut s, round_key(8 * m as usize + j), &mp);
}
s
};
for t in [0u32, 1, 0x0fff_ffff, 0xffff_ffff] {
assert_eq!(derive_item(t, &mp, &cache), by_hand(t), "m {m} item {t}");
}
}
// the interpreter's fetch path at the genesis cache: words 0, 15, 16 and MASK of a 2^20-word dataset agree with
// the item derivation, under v3
let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape::for_class(&V3_CLASS));
let m = ds.memhard().unwrap();
for w in [0u32, 15, 16, 17, ds.mask - 1, ds.mask] {
assert_eq!(ds.word(w), derive_item(w >> 4, &m.params, &m.cache)[(w & 15) as usize]);
assert_eq!(ds.word(w), m.word(w));
}
// a load at an out-of-range register masks to the dataset: the word at mask + 1 is the word at 0
assert_eq!(ds.word(ds.mask.wrapping_add(1)), ds.word(0));
}
/// Two independent epochs of the same seed and day: every vector and every emitted file identical; the pinned v3
/// pack is what a third export writes.
#[test]
fn determinism_v3() {
let build = || Epoch {
program: generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, None),
dataset: DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 28, Shape::for_class(&V3_CLASS)),
};
let a = build();
let b = build();
let pa = export_pack(&a, DAY, "a");
let pb = export_pack(&b, DAY, "a");
assert_eq!(pa.outs, pb.outs);
assert_eq!(pa.vectors, pb.vectors);
assert_eq!(pa.files, pb.files);
for (w, warp) in [(0u32, 0usize), (4096, 1), (1_000_000, 2)] {
assert_eq!(a.hash_warp(w), pa.outs[warp]);
}
let dir = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-mixer/mx8-genesis");
for (name, text) in &pa.files {
if name == "vectors.json" || name == "vectors.h" {
continue; // the source string differs ("a" here)
}
let on_disk = std::fs::read_to_string(dir.join(name)).unwrap();
assert_eq!(&on_disk, text, "{name}");
}
}

View file

@ -5,28 +5,52 @@
//!
//! Packs: igneum-genesis-mh and igneum-devnet-v4-epoch0 (memory-hard; the latter from the devnet genesis hash as
//! the epoch seed and the day bytes of 2026-10-04), igneum-genesis and igneum-hourly (closed-form dataset,
//! interpreter regression only).
//! interpreter regression only); and, under `proto-cuda/packs-ca2-mixer/`, the class v3 packs mx8-genesis and
//! mx8-devnet-epoch0 (Counter ASIC 2.0, 5 October 2026: generator 3 on `V3_CLASS` = mixer x8 with the cache growth
//! rule, decided 22:05 UTC under the delegated rule; the same seeds and days as the two memory-hard v2 packs, so the
//! v2 program and cache carry over and only the dataset words and the hashes change) and the x4 candidate's packs
//! mx4-genesis and mx4-devnet-epoch0 (generator 2 with the load class in the id, the record of the x4 rows).
use igneum_pow::accept;
use igneum_pow::emit::{
cuda_kernel, cuda_kernel_bound, cuda_memhard_header, export_pack, metal_memhard, metal_program,
cuda_kernel, cuda_kernel_bound, cuda_memhard_header, export_pack, metal_memhard, metal_memhard_for, metal_program,
metal_program_bound, opencl_kernel, opencl_kernel_bound, program_header, program_json, LoadSource,
};
use igneum_pow::generator::{generate_from_seed_bytes, Op, GENERATOR_VERSION, LOAD_SLOTS};
use igneum_pow::memhard::CACHE_WORDS;
use igneum_pow::generator::{generate_from_seed_bytes, generate_from_seed_bytes_class, generate_from_seed_bytes_program_class, LoadClass, Op, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, LOAD_SLOTS, V3_CLASS};
use igneum_pow::memhard::{Shape, CACHE_WORDS};
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch};
use serde_json::Value;
use std::path::PathBuf;
use std::sync::OnceLock;
const PACKS: [&str; 4] = ["igneum-genesis-mh", "igneum-devnet-v4-epoch0", "igneum-genesis", "igneum-hourly"];
const PACKS: [&str; 8] = [
"igneum-genesis-mh",
"igneum-devnet-v4-epoch0",
"igneum-genesis",
"igneum-hourly",
"mx8-genesis",
"mx8-devnet-epoch0",
"mx4-genesis",
"mx4-devnet-epoch0",
];
/// The class v3 packs (generator 3 through the seam).
const PACKS_V3: [&str; 2] = ["mx8-genesis", "mx8-devnet-epoch0"];
fn packs_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs")
}
/// The directory a pack lives in: the class v3 packs under packs-ca2-mixer, the rest under packs.
fn pack_dir(pack: &str) -> PathBuf {
if pack.starts_with("mx4-") || pack.starts_with("mx8-") {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-mixer").join(pack)
} else {
packs_dir().join(pack)
}
}
fn read(pack: &str, file: &str) -> String {
let p = packs_dir().join(pack).join(file);
let p = pack_dir(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
@ -62,9 +86,22 @@ fn epoch(pack: &str) -> &'static Epoch {
_ => DatasetMode::ClosedForm,
};
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
let program = generate_from_seed_bytes(seed, &seed_bytes);
// a class v3 pack: generator 3 on V3_CLASS through the seam, the era bytes it records, the dataset
// in the class's shape on day 0 (the growth rule's genesis cache: every pinned pack is a day-0 size)
let program = match j["generator"].as_u64().unwrap() as u32 {
GENERATOR_VERSION_V3 => {
let era = j.get("era_seed_bytes").map(unhex);
generate_from_seed_bytes_program_class(seed, &seed_bytes, ProgramClass::V3, era.as_deref())
}
// a generator 2 pack with a load class (the x4 candidate's packs): the class from program.json
_ => match j.get("load_class").and_then(|c| c.as_str()) {
Some(c) => generate_from_seed_bytes_class(seed, &seed_bytes, LoadClass::parse(c).expect("a load class name")),
None => generate_from_seed_bytes(seed, &seed_bytes),
},
};
let shape = Shape::for_class(&program.class);
let mut dataset =
DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), mode, log2);
DatasetSource::from_key_shape(igneum_pow::seed::seed_words_from_bytes(&day_bytes), mode, log2, shape);
dataset.key_bytes = day_bytes;
(p.to_string(), Epoch { program, dataset })
})
@ -81,7 +118,8 @@ fn check_program_json(pack: &str) {
let j = json(pack, "program.json");
let p = &epoch(pack).program;
assert_eq!(j["format"].as_str().unwrap(), "igneum-program-pack-3");
assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION, "{pack}: generator version");
assert_eq!(j["generator"].as_u64().unwrap() as u32, p.generator, "{pack}: generator version");
assert_eq!(p.generator, if PACKS_V3.contains(&pack) { GENERATOR_VERSION_V3 } else { GENERATOR_VERSION });
assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt, "{pack}: attempt");
assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id");
let sw: Vec<u32> = j["seed_words"].as_array().unwrap().iter().map(hex32).collect();
@ -129,7 +167,7 @@ fn genesis_program_shape() {
#[test]
fn mixer_params_match_pack() {
for pack in ["igneum-genesis-mh", "igneum-devnet-v4-epoch0"] {
for pack in ["igneum-genesis-mh", "igneum-devnet-v4-epoch0", "mx8-genesis", "mx8-devnet-epoch0", "mx4-genesis", "mx4-devnet-epoch0"] {
let j = json(pack, "program.json");
let mp = &epoch(pack).dataset.memhard().unwrap().params;
let key: Vec<u32> = j["dataset"]["key"].as_array().unwrap().iter().map(hex32).collect();
@ -166,19 +204,21 @@ fn cache_matches_vectors() {
fn check_dataset_words(pack: &str) {
let v = json(pack, "vectors.json");
let ds = &epoch(pack).dataset;
let e = epoch(pack);
let ds = &e.dataset;
// a pack's self-test words are read under its program's layout (the era layout; linear for every v2 pack)
let head: Vec<u32> = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect();
for (i, h) in head.iter().enumerate() {
assert_eq!(ds.word(i as u32), *h, "{pack}: dataset[{i}]");
assert_eq!(e.dataset_word(i as u32), *h, "{pack}: dataset[{i}]");
}
let last_index = v["dataset_last_index"].as_u64().unwrap() as u32;
assert_eq!(last_index, ds.mask);
assert_eq!(ds.word(last_index), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]");
assert_eq!(e.dataset_word(last_index), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]");
let samples = v["dataset_samples"].as_array().unwrap();
assert_eq!(samples.len(), 64);
for s in samples {
let idx = s["index"].as_u64().unwrap() as u32;
assert_eq!(ds.word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]");
assert_eq!(e.dataset_word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]");
}
}
@ -261,12 +301,15 @@ fn check_sources(pack: &str) {
assert_same_text(pack, "program.h", &program_header(p, &day, &e.dataset));
if let Some(mp) = mp {
assert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp));
assert_same_text(pack, "memhard.metal", &metal_memhard(mp));
assert_same_text(pack, "memhard.metal", &igneum_pow::emit::metal_memhard_layout(mp, p.class.layout()));
}
let got = program_json(p, &day, &e.dataset);
assert_same_text(pack, "program.json", &got);
let _: Value = serde_json::from_str(&got).expect("program.json is valid JSON");
// Every load in every emitted hash kernel has the masked form, and there are exactly 16 of them.
// Every load in every emitted hash kernel has the masked form, and there are exactly 16 of them (a class with
// the era layout inside has the era form instead: `((rotl_imm(rN * M, R) & WM) | OFF) & mask`, checked by
// era_emitted_sources_match_and_loads_have_the_era_form over the era packs, and here by the same count).
let era_load = if p.class.era.is_some() { "((rotl_imm(r" } else { "" };
for (file, load, masked) in [
("kernel.cu", "ds[r", " & mask]"),
("kernel_bound.cu", "ds[r", " & mask]"),
@ -274,6 +317,7 @@ fn check_sources(pack: &str) {
("program_bound.metal", "dataset[r", " & MASK]"),
] {
let text = read(pack, file);
let load = if era_load.is_empty() { load } else { era_load };
assert_eq!(text.matches(load).count(), LOAD_SLOTS, "{pack}/{file}: 16 loads");
assert_eq!(text.matches(masked).count(), LOAD_SLOTS, "{pack}/{file}: 16 masked loads");
}
@ -287,6 +331,85 @@ fn emitted_sources_match_all_packs() {
}
/// The whole pack as `export` writes it: vectors.json and vectors.h match, and the file list is the full set.
/// The class v3 packs (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): generator 3 on V3_CLASS = mx8; the program of
/// each is the v2 program of the same seed instruction for instruction (v2 loads take no width roll); the cache is
/// the v2 cache (day 0 of the growth rule: 2^26 words, the same FNV-1a 64); the dataset words differ from v2's;
/// program.json, program.h and the emitted memhard core say so; the id carries generator 3.
#[test]
fn v3_packs_are_the_v2_seeds_under_mixer_x8() {
assert_eq!(V3_CLASS.name(), "mx8");
assert_eq!(V3_CLASS.mixer_mult, 8);
assert!(V3_CLASS.growth);
assert_eq!(LoadClass { mixer_mult: 4, ..V3_CLASS }, LoadClass::MX4, "the x4 candidate differs from v3 in the multiplier alone");
for (v3, v2) in [("mx8-genesis", "igneum-genesis-mh"), ("mx8-devnet-epoch0", "igneum-devnet-v4-epoch0")] {
let e3 = epoch(v3);
let e2 = epoch(v2);
let j = json(v3, "program.json");
assert_eq!(j["program_class"].as_str().unwrap(), "v3");
assert!(j["load_class"].as_str().unwrap().starts_with("mx8"), "{v3}: mx8, or mx8 with the era inside");
assert_eq!(j["mixer_mult"].as_u64().unwrap(), 8);
assert_eq!(j["cache_growth"].as_bool().unwrap(), true);
assert_eq!(j["dataset"]["mixer_mult"].as_u64().unwrap(), 8);
assert_eq!(j["dataset"]["cache"]["log2_words"].as_u64().unwrap(), 26);
assert_eq!(e3.program.generator, GENERATOR_VERSION_V3);
if e3.program.era_bytes.is_some() {
// the era layout composed into class v3 (docs/plans/era-layout.md, 5 October 2026): a chain pack carries an
// era, so its class is V3_CLASS with the era drawn inside and its stream takes two window draws per
// instruction; the v2 program carries over only in the seed, the attempt and the day
assert_eq!(LoadClass { era: None, ..e3.program.class }, V3_CLASS, "{v3}: the composed class");
assert!(e3.program.class.era.is_some());
assert_ne!(e3.program.instrs, e2.program.instrs, "{v3}: the era windows change the stream");
} else {
assert_eq!(e3.program.class, V3_CLASS);
assert_eq!(e3.program.instrs, e2.program.instrs, "{v3}: the v2 program under the v3 construction");
}
assert_eq!(e3.program.seed, e2.program.seed);
assert_eq!(e3.program.attempt, e2.program.attempt);
assert_ne!(e3.program.program_id(), e2.program.program_id());
assert_eq!(e3.program.program_id(), igneum_pow::generator::program_id(GENERATOR_VERSION_V3, &e3.program.seed, e3.program.attempt));
let m3 = e3.dataset.memhard().unwrap();
let m2 = e2.dataset.memhard().unwrap();
assert_eq!(m3.shape(), Shape { mixer_mult: 8, cache_log2_words: 26 });
assert_eq!(m3.cache.fnv1a64(), m2.cache.fnv1a64(), "{v3}: the same cache as v2 on day 0");
assert_eq!(m3.params.rot, m2.params.rot);
assert_eq!(e3.dataset.log2_words, 28);
assert_ne!(e3.dataset.word(0), e2.dataset.word(0), "{v3}: the dataset words differ");
assert_ne!(e3.hash_warp(0), e2.hash_warp(0));
let h = read(v3, "program.h");
assert!(h.contains("#define IGNEUM_GENERATOR 3\n"));
assert!(h.contains("#define IGNEUM_PROGRAM_CLASS \"v3\"\n"));
assert!(h.contains("#define IGNEUM_MIXER_MULT 8"));
assert!(h.contains("#define IGNEUM_CACHE_GROWTH 1"));
assert!(h.contains("#define IGNEUM_CACHE_LOG2_WORDS 26\n"));
assert!(h.contains("#define IGNEUM_LOAD_CLASS \"mx8"), "mx8, or mx8 with the era inside");
for file in ["memhard.h", "memhard.metal", "kernel.cl"] {
let text = read(v3, file);
assert_eq!(text.matches("j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u))").count(), 1, "{v3}/{file}");
assert_eq!(text.matches("j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u))").count(), 1, "{v3}/{file}");
}
for file in ["memhard.h", "memhard.metal", "kernel.cl"] {
let text = read(v2, file);
assert_eq!(text.matches("j < 8u").count(), 0, "{v2}/{file}: the v2 text has no multiplier loop");
}
}
// the devnet v3 pack records era 0's stand-in, the devnet genesis hash
let j = json("mx8-devnet-epoch0", "program.json");
assert_eq!(unhex(&j["era_seed_bytes"]), unhex(&j["seed_bytes"]));
assert!(read("mx8-devnet-epoch0", "program.h").contains("#define IGNEUM_ERA_SEED_HEX \"edc4fa844da9dc98"));
assert!(json("mx8-genesis", "program.json").get("era_seed_bytes").is_none());
// the x4 candidate's packs: generator 2, the class in the id, the v2 program of the seed, mixer x4
for (x4, v2) in [("mx4-genesis", "igneum-genesis-mh"), ("mx4-devnet-epoch0", "igneum-devnet-v4-epoch0")] {
let e4 = epoch(x4);
let j = json(x4, "program.json");
assert_eq!(e4.program.generator, GENERATOR_VERSION);
assert_eq!(j["load_class"].as_str().unwrap(), "mx4");
assert_eq!(e4.program.class, LoadClass::MX4);
assert_eq!(e4.program.instrs, epoch(v2).program.instrs);
assert_eq!(e4.dataset.memhard().unwrap().shape(), Shape { mixer_mult: 4, cache_log2_words: 26 });
assert!(read(x4, "memhard.h").contains("j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u))"));
}
}
fn check_export(pack: &str) {
let e = epoch(pack);
let v = json(pack, "vectors.json");
@ -311,7 +434,7 @@ fn check_export(pack: &str) {
expected.extend(["memhard.h", "memhard.metal"]);
}
assert_eq!(out.files.iter().map(|(n, _)| n.as_str()).collect::<Vec<_>>(), expected);
let mut on_disk: Vec<String> = std::fs::read_dir(packs_dir().join(pack))
let mut on_disk: Vec<String> = std::fs::read_dir(pack_dir(pack))
.unwrap()
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
.filter(|n| !n.starts_with('.'))
@ -350,3 +473,429 @@ fn devnet_pack_is_the_chain_derivation() {
assert_eq!(e.program.instrs, epoch("igneum-devnet-v4-epoch0").program.instrs);
assert_eq!(e.hash_warp(0), epoch("igneum-devnet-v4-epoch0").hash_warp(0));
}
// ---------------------------------------------------------------------------------------------------------
// Era layout packs (5 October 2026, docs/plans/era-layout.md): proto-cuda/packs-ca2-era/era-<n>, n in 0..5, the
// devnet epoch seed and day bytes under the era class of test era seed igneum-era-test/<n>, the width pinned at
// 4 bytes (allowed_widths in program.json). Checked like the pinned packs, plus the one load form of 1.3 by text search.
// ---------------------------------------------------------------------------------------------------------
use igneum_pow::generator::{generate_era, EraParams, V3_ALLOWED};
const ERA_PACKS: [&str; 6] = ["era-0", "era-1", "era-2", "era-3", "era-4", "era-5"];
fn era_packs_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-era")
}
fn era_read(pack: &str, file: &str) -> String {
let p = era_packs_dir().join(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
fn era_json(pack: &str, file: &str) -> Value {
serde_json::from_str(&era_read(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"))
}
/// The era class of a pack: the era bytes from program.json (`era_seed_bytes`, the chain's `E_n`; the test seed
/// `igneum-era-test/<n>` of the pack's number gives the same bytes), the allowed set from `era.allowed_widths`
/// (class v3's `V3_ALLOWED`); the pack's recorded stream words and draw must be the class's.
fn era_class(pack: &str) -> LoadClass {
let j = era_json(pack, "program.json");
let n: u64 = pack.trim_start_matches("era-").parse().unwrap();
let eb = unhex(&j["era_seed_bytes"]);
assert_eq!(eb, EraParams::test_era_bytes(&format!("igneum-era-test/{n}")).to_vec(), "{pack}: the era bytes of test seed {n}");
let allowed: Vec<u8> = j["era"]["allowed_widths"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u8).collect();
assert_eq!(allowed, V3_ALLOWED.to_vec(), "{pack}: class v3's width set");
// the measurement packs of 5 October 2026: the era layout over version 2's construction (mixer x1, the genesis
// cache), generator 3 and the era bytes recorded; the chain's class v3 composes the same draw over LoadClass::MX4
// (generator tests era_programs_are_accepted and program_classes), and the integration re-exports these packs
let c = LoadClass::era(LoadClass::V2, &eb, &allowed);
let e = c.era.unwrap();
let words: Vec<u32> = j["era"]["seed_words"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(e.words.to_vec(), words, "{pack}: era seed words");
assert_eq!(e.width_words as u64, j["era"]["width_words"].as_u64().unwrap(), "{pack}: width");
assert_eq!(e.stride_mul, hex32(&j["era"]["stride_mul"]), "{pack}: stride mul");
assert_eq!(e.stride_rot as u64, j["era"]["stride_rot"].as_u64().unwrap(), "{pack}: stride rot");
let pos: Vec<u8> = j["era"]["interleave"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u8).collect();
assert_eq!(e.pos.to_vec(), pos, "{pack}: interleave");
c
}
fn era_epoch(pack: &str) -> &'static Epoch {
static E: OnceLock<Vec<(String, Epoch)>> = OnceLock::new();
let all = E.get_or_init(|| {
ERA_PACKS
.iter()
.map(|p| {
let j = era_json(p, "program.json");
let seed = j["seed"].as_str().unwrap();
let seed_bytes = unhex(&j["seed_bytes"]);
let day_bytes = unhex(&j["dataset"]["day_bytes"]);
assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard");
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
let class = era_class(p);
assert_eq!(log2, igneum_pow::verify::DEFAULT_DATASET_LOG2);
let eb = unhex(&j["era_seed_bytes"]);
let program = generate_era(seed, &seed_bytes, LoadClass::V2, &eb, &V3_ALLOWED);
assert_eq!(program.class, class, "{p}: the pack's era class");
assert_eq!(program.generator, GENERATOR_VERSION_V3);
assert_eq!(program.era_bytes.as_deref(), Some(&eb[..]));
let mut dataset = DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2);
dataset.key_bytes = day_bytes;
(p.to_string(), Epoch { program, dataset })
})
.collect()
});
&all.iter().find(|(n, _)| n == pack).unwrap().1
}
/// program.json of an era pack: generator, attempt, id, class, every instruction with width, win and off, and the
/// program passes the acceptance rule.
#[test]
fn era_program_json_matches() {
for pack in ERA_PACKS {
let j = era_json(pack, "program.json");
let p = &era_epoch(pack).program;
assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION_V3, "{pack}: a class v3 pack");
assert_eq!(j["program_class"].as_str().unwrap(), "v3");
assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt, "{pack}: attempt");
assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id");
assert_eq!(j["load_class"].as_str().unwrap(), p.class.name(), "{pack}: class");
assert_eq!(j["bytes_per_hash"].as_u64().unwrap() as usize, p.bytes_per_hash());
assert_eq!(p.loads_per_hash(), 8 * LOAD_SLOTS);
assert!(accept::check(p).is_ok(), "{pack}: acceptance");
let instrs = j["instructions"].as_array().unwrap();
assert_eq!(instrs.len(), p.instrs.len());
for (k, (ins, ji)) in p.instrs.iter().zip(instrs).enumerate() {
assert_eq!(Op::from_name(ji["op"].as_str().unwrap()).unwrap(), ins.op, "{pack} #{k} op");
assert_eq!(ji["dst"].as_u64().unwrap(), ins.dst as u64);
assert_eq!(ji["src"].as_u64().unwrap(), ins.src as u64);
assert_eq!(hex32(&ji["imm"]), ins.imm);
assert_eq!(ji["width"].as_u64().unwrap(), ins.width as u64, "{pack} #{k} width");
assert_eq!(ji["win"].as_u64().unwrap(), ins.win as u64, "{pack} #{k} win");
assert_eq!(ji["off"].as_u64().unwrap(), ins.off as u64, "{pack} #{k} off");
if ins.op == Op::Load {
assert_eq!(ins.width, p.class.era.unwrap().width_words);
assert!(ins.win <= 2 && (ins.off as u32) < (1u32 << ins.win));
}
}
}
}
/// The six era packs are the same program seed under six draws: the instruction lists agree, the widths and layouts
/// follow the draw, and the dataset words differ from the linear layout exactly when the interleave is not linear.
#[test]
fn era_packs_share_the_program_and_differ_in_layout() {
let linear = &epoch("igneum-devnet-v4-epoch0").dataset;
for pack in ERA_PACKS {
let e = era_epoch(pack);
assert_eq!(e.program.seed_bytes, epoch("igneum-devnet-v4-epoch0").program.seed_bytes, "{pack}: the devnet seed");
let strip = |p: &igneum_pow::generator::Program| {
p.instrs.iter().map(|i| (i.op, i.dst, i.src, i.src2, i.imm, i.imm2, i.rot, i.bit, i.mask, i.win, i.off)).collect::<Vec<_>>()
};
assert_eq!(strip(&e.program), strip(&era_epoch("era-0").program), "{pack}: same stream as era-0");
let l = e.program.class.layout();
let same_at_1 = (0..64u32).all(|w| e.dataset_word(w * 977 + 1) == linear.word(w * 977 + 1));
assert_eq!(same_at_1, l.is_linear(), "{pack}: layout {:?}", l.pos);
assert_eq!(e.dataset_word(0), linear.word(0), "{pack}: word 0 is item 0 word 0 in every layout");
// the chain's shared day cache serves every era: the day's dataset source is the pinned pack's, bit for bit
assert_eq!(e.dataset.key, linear.key);
assert_eq!(e.dataset.memhard().unwrap().cache.fnv1a64(), linear.memhard().unwrap().cache.fnv1a64());
}
}
#[test]
fn era_dataset_words_and_vectors_match() {
for pack in ERA_PACKS {
let v = era_json(pack, "vectors.json");
let e = era_epoch(pack);
let ds = &e.dataset;
let head: Vec<u32> = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect();
for (i, h) in head.iter().enumerate() {
assert_eq!(e.dataset_word(i as u32), *h, "{pack}: dataset[{i}]");
}
assert_eq!(e.dataset_word(ds.mask), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]");
for s in v["dataset_samples"].as_array().unwrap() {
let idx = s["index"].as_u64().unwrap() as u32;
assert_eq!(e.dataset_word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]");
}
assert_eq!(ds.memhard().unwrap().cache.fnv1a64(), hex64(&v["cache_fnv1a64"]));
let mut n = 0;
for w in v["warps"].as_array().unwrap() {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let expected: Vec<u64> = w["expected"].as_array().unwrap().iter().map(hex64).collect();
let got = e.hash_warp(base);
for lane in 0..32 {
assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}");
n += 1;
}
assert_eq!(e.hash(base + 7), expected[7]);
}
assert_eq!(n, 96, "{pack}");
}
}
/// Every emitted file of every era pack matches the emitters byte for byte, the export reproduces vectors.json and
/// vectors.h, and every dataset load in every hash kernel has the one era form (no plain `ds[rN & mask]` remains).
#[test]
fn era_emitted_sources_match_and_loads_have_the_era_form() {
for pack in ERA_PACKS {
let e = era_epoch(pack);
let day = era_json(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string();
let source = era_json(pack, "vectors.json")["source"].as_str().unwrap().to_string();
let out = export_pack(e, &day, &source);
for (name, text) in &out.files {
let want = era_read(pack, name);
assert!(text == &want, "{pack}/{name} differs from the emitter");
}
let mut on_disk: Vec<String> = std::fs::read_dir(era_packs_dir().join(pack))
.unwrap()
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
.filter(|n| !n.starts_with('.') && n != "seeds.txt")
.collect();
on_disk.sort();
let mut want: Vec<String> = out.files.iter().map(|(n, _)| n.clone()).collect();
want.sort();
assert_eq!(on_disk, want, "{pack}: the pack holds the export's files and seeds.txt only");
let era = e.program.class.era.unwrap();
let mul = format!("0x{:08x}u", era.stride_mul);
for (file, mask) in [
("kernel.cu", "mask"),
("kernel_bound.cu", "mask"),
("kernel.cl", "mask"),
("kernel_bound.cl", "mask"),
("program.metal", "MASK"),
("program_bound.metal", "MASK"),
] {
let text = era_read(pack, file);
let kernels = if file.starts_with("kernel_bound") || file == "kernel.cu" || file == "kernel.cl" || file.starts_with("program") { 1 } else { 1 };
// kernel_bound.cl carries igneum_hash and igneum_hash_bound: two kernels
let kernels = if file == "kernel_bound.cl" { 2 } else { kernels };
let era_form: usize = text
.lines()
.filter(|l| l.contains("rotl_imm(r") && l.contains(&format!(" * {mul}, {}u) & ", era.stride_rot)) && l.contains(&format!(") & {mask}")))
.filter(|l| l.contains("ds[") || l.contains("dataset[") || l.contains("b_ = "))
.count();
assert_eq!(era_form, LOAD_SLOTS * kernels, "{pack}/{file}: {} loads of the era form", LOAD_SLOTS * kernels);
let plain = text.lines().filter(|l| l.contains("ds[r") || l.contains("dataset[r")).count();
assert_eq!(plain, 0, "{pack}/{file}: a load without the era form");
}
// the layout helpers appear exactly when the layout is not linear
let mh = era_read(pack, "memhard.h");
assert_eq!(mh.contains("mh_addr("), !era.layout().is_linear(), "{pack}: memhard.h layout helpers");
}
}
/// An era pack's dataset is a prefix at every size of at least 2^16 words: the 2^20-word source gives the pack's
/// words below 2^20.
#[test]
fn era_dataset_is_a_prefix_at_smaller_sizes() {
for pack in ["era-1", "era-3"] {
let e = era_epoch(pack);
let small = DatasetSource::from_key(e.dataset.key, DatasetMode::MemoryHard, 20);
let l = e.program.class.layout();
for w in [0u32, 1, 2, 3, 16, 255, 4096, 65_535, 65_536, 0x000f_ffff] {
assert_eq!(small.word_at(l, w), e.dataset_word(w), "{pack}: w {w}");
}
}
}
// ---------------------------------------------------------------------------------------------------------
// Hot-table experiment (5 October 2026, docs/plans/hot-table.md): the five packs under proto-cuda/packs-ca2-hot/ are
// pinned the same way (program, vectors, every emitted file byte for byte), plus the hot table's fingerprint and the
// one-form load check: exactly 16 - k masked dataset loads and exactly k hot loads in every hash kernel.
// ---------------------------------------------------------------------------------------------------------
const HOT_PACKS: [&str; 8] = ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "hot32k4a", "hot64k4a", "hot96k4a"];
fn hot_packs_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-hot")
}
fn hread(pack: &str, file: &str) -> String {
let p = hot_packs_dir().join(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
fn hjson(pack: &str, file: &str) -> Value {
serde_json::from_str(&hread(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"))
}
/// The epoch of a hot pack from program.json alone: the class from `load_class`, the program from the seed bytes,
/// the dataset from the day bytes, the hot table from the seed bytes (what `Epoch::from_seed_bytes_class` does).
fn hepoch(pack: &str) -> &'static Epoch {
static E: OnceLock<Vec<(String, Epoch)>> = OnceLock::new();
let all = E.get_or_init(|| {
HOT_PACKS
.iter()
.map(|p| {
let j = hjson(p, "program.json");
let seed = j["seed"].as_str().unwrap();
let seed_bytes = unhex(&j["seed_bytes"]);
let day_bytes = unhex(&j["dataset"]["day_bytes"]);
assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard");
let class = LoadClass::parse(j["load_class"].as_str().unwrap()).unwrap();
assert_eq!(class.name(), *p, "the pack directory is the class name");
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
let program = generate_from_seed_bytes_class(seed, &seed_bytes, class);
let mut dataset =
DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2);
dataset.key_bytes = day_bytes;
dataset.attach_hot_for(&program);
(p.to_string(), Epoch { program, dataset })
})
.collect()
});
&all.iter().find(|(n, _)| n == pack).unwrap().1
}
fn hassert_same_text(pack: &str, file: &str, got: &str) {
let want = hread(pack, file);
if got != want {
let (gl, wl): (Vec<&str>, Vec<&str>) = (got.lines().collect(), want.lines().collect());
for i in 0..gl.len().max(wl.len()) {
let g = gl.get(i).copied().unwrap_or("<eof>");
let w = wl.get(i).copied().unwrap_or("<eof>");
if g != w {
panic!("{pack}/{file} differs at line {}:\n pack: {w}\n rust: {g}", i + 1);
}
}
panic!("{pack}/{file} differs only in trailing bytes (len {} vs {})", got.len(), want.len());
}
}
#[test]
fn hot_packs_program_and_vectors() {
for pack in HOT_PACKS {
let e = hepoch(pack);
let p = &e.program;
let j = hjson(pack, "program.json");
let h = p.class.hot.unwrap();
assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION);
assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt);
assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id");
let dataset_slots = if h.added { 16 } else { 16 - h.k as usize };
assert_eq!(j["loads_per_hash"].as_u64().unwrap() as usize, (dataset_slots + h.k as usize) * 8);
assert_eq!(j["hot_table"]["mb"].as_u64().unwrap(), h.mb as u64);
assert_eq!(j["hot_table"]["slots"].as_u64().unwrap(), h.k as u64);
assert_eq!(j["hot_table"]["dataset_slots"].as_u64().unwrap() as usize, dataset_slots);
assert_eq!(j["hot_table"]["words"].as_u64().unwrap() as u32, p.hot_words());
assert_eq!(j["op_mix"]["hot"].as_u64().unwrap(), h.k as u64, "{pack}: k hot instructions");
assert_eq!(j["op_mix"]["load"].as_u64().unwrap() as usize, dataset_slots);
assert_eq!(p.items_per_warp(), dataset_slots * 8 * 32);
assert!(accept::check(p).is_ok(), "{pack}: passes the acceptance rule");
let v2 = &epoch("igneum-genesis-mh").program;
if !h.added {
// replaced form: the version 2 genesis program with k loads redirected (attempt 0 on both)
assert_eq!(p.attempt, v2.attempt);
for (a, b) in p.instrs.iter().zip(v2.instrs.iter()) {
if a.op == Op::Hot {
assert_eq!(b.op, Op::Load);
} else {
assert_eq!(a, b);
}
}
} else {
// added form: 16 + k load slots, so another slot draw and another program; 16 dataset loads stay
assert_ne!(p.instrs, v2.instrs);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16);
assert!(p.instrs.iter().all(|i| i.width == 1));
}
// the hot table: the pack's head, last line and fingerprint
let v = hjson(pack, "vectors.json");
let t = e.dataset.hot.as_ref().unwrap();
assert_eq!(t.n_words(), p.hot_words());
let head: Vec<u32> = v["hot_head"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(&t.words()[..16], &head[..]);
let last: Vec<u32> = v["hot_last_line"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(&t.words()[t.words().len() - 16..], &last[..]);
assert_eq!(t.fnv1a64(), hex64(&v["hot_fnv1a64"]), "{pack}: hot_fnv1a64");
assert_eq!(t.key, igneum_pow::memhard::hot_key(&p.seed_bytes));
// the cache is the day's, unchanged by the class
assert_eq!(e.dataset.memhard().unwrap().cache.fnv1a64(), 0x48c4f5bf24166b2e);
// 96 vectors
let warps = v["warps"].as_array().unwrap();
assert_eq!(warps.len(), 3);
for w in warps {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let expected: Vec<u64> = w["expected"].as_array().unwrap().iter().map(hex64).collect();
let got = e.hash_warp(base);
for lane in 0..32 {
assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}");
}
assert_eq!(e.hash(base + 5), expected[5]);
}
// the dataset words are the day's
let head: Vec<u32> = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect();
for (i, hd) in head.iter().enumerate() {
assert_eq!(e.dataset.word(i as u32), *hd);
}
}
// the same k at three sizes: identical programs, three fingerprints, three vector sets
let a = hepoch("hot32k4");
let b = hepoch("hot64k4");
let c = hepoch("hot96k4");
assert_eq!(a.program.instrs, b.program.instrs);
assert_eq!(b.program.instrs, c.program.instrs);
assert_ne!(a.hash_warp(0), b.hash_warp(0));
assert_ne!(b.hash_warp(0), c.hash_warp(0));
}
#[test]
fn hot_packs_emitted_sources_and_load_forms() {
for pack in HOT_PACKS {
let e = hepoch(pack);
let p = &e.program;
let k = p.class.hot.unwrap().k as usize;
let dataset_loads = p.class.dataset_slots();
let day = hjson(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string();
let mp = &e.dataset.memhard().unwrap().params;
hassert_same_text(pack, "kernel.cu", &cuda_kernel(p, Some(mp)));
hassert_same_text(pack, "kernel_bound.cu", &cuda_kernel_bound(p, Some(mp)));
hassert_same_text(pack, "program.metal", &metal_program(p, e.dataset.log2_words, LoadSource::Stored));
hassert_same_text(pack, "program_bound.metal", &metal_program_bound(p, e.dataset.log2_words));
hassert_same_text(pack, "kernel.cl", &opencl_kernel(p, Some(mp)));
hassert_same_text(pack, "kernel_bound.cl", &opencl_kernel_bound(p, Some(mp)));
hassert_same_text(pack, "program.h", &program_header(p, &day, &e.dataset));
hassert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp));
hassert_same_text(pack, "memhard.metal", &metal_memhard_for(p, mp));
assert_ne!(metal_memhard_for(p, mp), metal_memhard(mp), "{pack}: the hot fill kernel is in memhard.metal");
let got = program_json(p, &day, &e.dataset);
hassert_same_text(pack, "program.json", &got);
let _: Value = serde_json::from_str(&got).expect("program.json is valid JSON");
let v = hjson(pack, "vectors.json");
let out = export_pack(e, &day, v["source"].as_str().unwrap());
let file = |name: &str| -> &str { &out.files.iter().find(|(n, _)| n == name).unwrap().1 };
hassert_same_text(pack, "vectors.json", file("vectors.json"));
hassert_same_text(pack, "vectors.h", file("vectors.h"));
assert_eq!(out.files.len(), 12);
// One form per dialect, exactly 16 - k masked dataset loads and k hot loads in every hash kernel; the fill
// kernel is present once per source that builds the table.
for (file, load, masked, hot) in [
("kernel.cu", "ds[r", " & mask]", "hot[__umulhi(r"),
("kernel_bound.cu", "ds[r", " & mask]", "hot[__umulhi(r"),
("program.metal", "dataset[r", " & MASK]", "hot[mulhi(r"),
("program_bound.metal", "dataset[r", " & MASK]", "hot[mulhi(r"),
("kernel.cl", "ds[r", " & mask]", "hot[mul_hi(r"),
] {
let text = hread(pack, file);
assert_eq!(text.matches(load).count(), dataset_loads, "{pack}/{file}: {dataset_loads} dataset loads");
assert_eq!(text.matches(masked).count(), dataset_loads, "{pack}/{file}: masked loads");
assert_eq!(text.matches(hot).count(), k, "{pack}/{file}: {k} hot loads");
assert!(text.contains(&format!("#define HOT_WORDS 0x{:08x}u", p.hot_words())), "{pack}/{file}: HOT_WORDS literal");
}
// kernel_bound.cl carries both kernels
let text = hread(pack, "kernel_bound.cl");
assert_eq!(text.matches("hot[mul_hi(r").count(), 2 * k);
assert_eq!(text.matches("ds[r").count(), 2 * dataset_loads);
for file in ["kernel.cu", "kernel.cl", "kernel_bound.cl", "memhard.metal"] {
assert_eq!(hread(pack, file).matches("igneum_hot_fill(").count(), 1, "{pack}/{file}: one hot fill kernel");
}
assert_eq!(hread(pack, "memhard.h").matches("void ht_segment(").count(), 1);
let ph = hread(pack, "program.h");
assert!(ph.contains(&format!("#define IGNEUM_HOT_MB {}", p.class.hot.unwrap().mb)));
assert!(ph.contains(&format!("#define IGNEUM_HOT_SLOTS {k}")));
assert!(ph.contains("igneum_launch_hot_fill("));
}
}

766
igneum-pow/tests/scratch.rs Normal file
View file

@ -0,0 +1,766 @@
//! Soundness tests of layer 3 of `docs/plans/counter-asic-2.md`: the per-warp scratch with read-modify-writes
//! (variant 5 of the read-width experiment, `LoadClass::scratch(k, kb)`). Analysis and results:
//! `docs/analysis/scratch-soundness.md`. Every test is parametric over the class's slot count
//! (`scratch_slots_per_lane()`), so the 32 and 128 KiB geometries and any later one run the same checks.
//!
//! What runs under plain `cargo test`:
//! 1. `rewrite_is_a_bijection_of_the_fold_value`, `fill_is_a_bijection_of_the_nonce`: the written words as
//! functions (question 1).
//! 2. `written_words_unbiased_and_rehit_rates`: bit bias of every written word over 2^11 units x 3 seeds per class
//! (the TESTS.md section 3 shape), and the measured slot re-hit rate against the birthday formula (question 2).
//! 3. `edge_programs_match_the_hand_model`: hand-built programs that drive every read-modify-write of a hash to
//! slot 0, slot MASK, through out-of-range registers, to one slot per lane, alternating two slots, and 16
//! read-modify-writes per iteration on one slot; the interpreter against an independent hand model, and the
//! hand model shown to have teeth (question 3, CPU half).
//! 4. `scr_packs_regenerate_and_pass_the_static_scratch_check`: every emitted kernel of every scr pack under
//! `proto-cuda/packs-readwidth` regenerates from its program.json and passes the static scratch-mask check;
//! the check is shown to fail on four deliberate breaks (question 4).
//! 5. `fuzz_scr_programs_cpu`: 200 generated scratch programs over the six classes, generator contract on every
//! instruction, 4 units each at base nonces across the 32-bit range including the wrap; with
//! `IGNEUM_SCRATCH_PACKS_OUT=<dir>` it also writes the packs (and the edge packs) for the Metal runs of
//! `proto-metal/packbench` (question 3 GPU half, question 4, `TESTS.md` section 9 shape).
use igneum_pow::emit::{
cuda_kernel, cuda_kernel_bound, export_pack, metal_program, metal_program_bound, opencl_kernel,
opencl_kernel_bound, vectors_json, LoadSource,
};
use igneum_pow::generator::{
generate_class, generate_from_seed_bytes_class, Instr, LoadClass, Op, Program, GENERATOR_VERSION, INSTR_COUNT,
ITERATIONS, LANES,
};
use igneum_pow::seed::{seed_words_from_bytes, SplitMix64};
use igneum_pow::verify::{
fold_words, interpret_warp_scratch, scratch_fill, scratch_rewrite, splitmix32, DatasetMode, DatasetSource,
Epoch, ScratchEvent, FOLD_MUL, FOLD_ROT,
};
use serde_json::Value;
use std::collections::HashMap;
use std::path::PathBuf;
/// The classes under study: the two capped geometries (32 and 128 KiB per warp: 64 and 256 slots per lane) at the
/// RMW shares the readwidth branch measures.
const CLASSES: [&str; 6] = ["scr2k32", "scr4k32", "scr8k32", "scr2k128", "scr4k128", "scr8k128"];
fn class(name: &str) -> LoadClass {
LoadClass::parse(name).unwrap_or_else(|| panic!("class {name}"))
}
// ---------------------------------------------------------------------------------------------------------------
// 1. The written words as functions (question 1)
// ---------------------------------------------------------------------------------------------------------------
/// For a fixed slot content `w`, each of the three rewritten words is a bijection of the fold value `x`
/// (`x ^ w1`, `rotl(x, 7) ^ w2`, `x + w0`), so the rewrite is injective in `x` and a uniform `x` gives a uniform
/// word in every position. Checked over 2^16 consecutive `x` for 16 random `w`.
#[test]
fn rewrite_is_a_bijection_of_the_fold_value() {
let mut rng = SplitMix64::new(0x7363_7261_7463_6801);
for _ in 0..16 {
let w = [rng.next() as u32, rng.next() as u32, rng.next() as u32];
let x0 = rng.next() as u32;
let mut seen = [vec![false; 1 << 16], vec![false; 1 << 16], vec![false; 1 << 16]];
for i in 0..(1u32 << 16) {
let x = x0.wrapping_add(i);
let out = scratch_rewrite(x, &w);
for j in 0..3 {
// a bijection of x maps 2^16 consecutive x to 2^16 distinct words; the low 16 bits alone are
// distinct for the xor words (x ^ c) and for the add word (x + c), since both act on the low 16
// bits as bijections of the low 16 bits of x; the rotl word is checked on its rotated-back bits
let key = if j == 1 { out[j].rotate_right(7) & 0xffff } else { out[j] & 0xffff };
assert!(!seen[j][key as usize], "word {j} repeats inside 2^16 consecutive x");
seen[j][key as usize] = true;
}
}
}
// The rewrite inverts: from the old content and any ONE written word the fold value is recovered, so a
// rewritten slot carries exactly 32 bits of new state (the point of question 2's arithmetic).
let w = [0x1234_5678, 0x9abc_def0, 0x0fed_cba9];
let x = 0xdead_beef;
let out = scratch_rewrite(x, &w);
assert_eq!(out[0] ^ w[1], x);
assert_eq!((out[1] ^ w[2]).rotate_right(7), x);
assert_eq!(out[2].wrapping_sub(w[0]), x);
}
/// For a fixed (seed, slot, j) the fill is a bijection of the lane nonce: `splitmix32` is a bijection of its
/// 32-bit input and the input `((base + lane) ^ s) + c` is a bijection of `base + lane`. Over 2^16 consecutive
/// nonces no fill word repeats, for 8 slots x 3 words.
#[test]
fn fill_is_a_bijection_of_the_nonce() {
let seed = seed_words_from_bytes(b"igneum-genesis");
for slot in [0u32, 1, 63, 64, 255, 1023, 2047] {
for j in 0..3u32 {
let mut words: Vec<u32> = (0..(1u32 << 16)).map(|n| scratch_fill(&seed, n, 0, slot, j)).collect();
words.sort_unstable();
words.dedup();
assert_eq!(words.len(), 1 << 16, "slot {slot} word {j}: fill words of 2^16 consecutive nonces are distinct");
}
}
// base + lane is the lane nonce: the fill of lane l at base b is the fill of lane 0 at base b + l
assert_eq!(scratch_fill(&seed, 0x1000, 7, 5, 2), scratch_fill(&seed, 0x1007, 0, 5, 2));
// and it wraps with the nonce: base 0xffffffe0, lane 31 is nonce 0xffffffff; lane 32 would be nonce 0
assert_eq!(scratch_fill(&seed, 0xffff_ffe0, 32, 5, 2), scratch_fill(&seed, 0, 0, 5, 2));
// the three word positions of one slot and nonce are three different permutation outputs
let f: Vec<u32> = (0..3).map(|j| scratch_fill(&seed, 12345, 7, 17, j)).collect();
assert!(f[0] != f[1] && f[1] != f[2] && f[0] != f[2]);
}
// ---------------------------------------------------------------------------------------------------------------
// 2. Uniformity of the written words and the slot re-hit rate (questions 1 and 2)
// ---------------------------------------------------------------------------------------------------------------
/// Birthday arithmetic: the expected number of distinct slots after `n` uniform draws from `s` slots.
fn expected_distinct(s: usize, n: usize) -> f64 {
let s = s as f64;
s * (1.0 - (1.0 - 1.0 / s).powi(n as i32))
}
struct ClassStats {
units: usize,
events: usize,
hits: usize,
/// ones count per bit of the written words, 3 x 32
ones: [[u64; 32]; 3],
/// ones count per bit of written XOR read (the change the rewrite makes to the slot)
delta_ones: [[u64; 32]; 3],
/// re-hit depth histogram: how many earlier RMWs the slot had seen in this unit (0 = first touch)
depth: Vec<usize>,
max_depth: usize,
/// how often each slot index was addressed (the slot comes from a register's low bits)
slot_hist: Vec<u64>,
}
fn class_stats(name: &str, seeds: &[&str], units_per_seed: usize) -> ClassStats {
let c = class(name);
let mut st = ClassStats {
units: 0,
events: 0,
hits: 0,
ones: [[0; 32]; 3],
delta_ones: [[0; 32]; 3],
depth: vec![0; 256],
max_depth: 0,
slot_hist: vec![0; c.scratch_slots_per_lane()],
};
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28);
for seed in seeds {
let p = generate_class(seed, c);
assert_eq!(p.scratch_ops_per_hash(), c.scratch_slots() * ITERATIONS);
for u in 0..units_per_seed {
let base = (u as u32).wrapping_mul(32).wrapping_add(0x4000_0000);
let (_, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true);
assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES);
let mut count: HashMap<(u8, u32), usize> = HashMap::new();
for e in &ev {
assert!(e.slot < c.scratch_slots_per_lane() as u32, "slot inside the lane's scratch");
let d = count.entry((e.lane, e.slot)).or_insert(0);
assert_eq!(e.hit, *d > 0, "hit flag agrees with the unit's own history");
assert_eq!(e.written, scratch_rewrite(e.x, &e.read));
if !e.hit {
let fill = [
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 0),
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 1),
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 2),
];
assert_eq!(e.read, fill, "a first touch reads the fill");
}
st.depth[(*d).min(255)] += 1;
st.max_depth = st.max_depth.max(*d);
st.slot_hist[e.slot as usize] += 1;
*d += 1;
st.events += 1;
st.hits += e.hit as usize;
for j in 0..3 {
for b in 0..32 {
st.ones[j][b] += ((e.written[j] >> b) & 1) as u64;
st.delta_ones[j][b] += (((e.written[j] ^ e.read[j]) >> b) & 1) as u64;
}
}
}
st.units += 1;
}
}
st
}
/// Bit bias of every written word (and of the change each rewrite makes) within 6 sigma of a fair coin, over
/// 3 seeds x 2^11 units per class (131,072 hashes per seed set); the slot re-hit rate against the birthday
/// formula within 3 percent relative. The table printed here is the one in the analysis.
#[test]
fn written_words_unbiased_and_rehit_rates() {
let seeds = ["igneum-genesis", "igneum-genesis/stats1", "igneum-genesis/stats2"];
let units = 1usize << 11;
println!("class | slots/lane | RMW/hash | events | re-hits | re-hit % | birthday % | slot chi2 z (spread) | max depth | max bias sigma | max delta bias sigma");
for name in CLASSES {
let c = class(name);
let st = class_stats(name, &seeds, units);
let n = st.events as f64;
let sigma = (n / 4.0).sqrt();
let mut worst = 0.0f64;
let mut worst_delta = 0.0f64;
for j in 0..3 {
for b in 0..32 {
let z = (st.ones[j][b] as f64 - n / 2.0).abs() / sigma;
let zd = (st.delta_ones[j][b] as f64 - n / 2.0).abs() / sigma;
assert!(z <= 6.0, "{name}: written word {j} bit {b} biased: {z:.2} sigma");
assert!(zd <= 6.0, "{name}: rewrite delta word {j} bit {b} biased: {zd:.2} sigma");
worst = worst.max(z);
worst_delta = worst_delta.max(zd);
}
}
let per_lane_hash = c.scratch_slots() * ITERATIONS;
let s = c.scratch_slots_per_lane();
let exp_hits = per_lane_hash as f64 - expected_distinct(s, per_lane_hash);
let exp_pct = 100.0 * exp_hits / per_lane_hash as f64;
let got_pct = 100.0 * st.hits as f64 / st.events as f64;
// chi-square of the slot histogram against uniform (df = s - 1): the slot is a register's low bits, and
// the measured re-hit rate runs above the uniform birthday rate (the finding of the analysis, question 2)
let expect_per_slot = n / s as f64;
let chi2: f64 = st.slot_hist.iter().map(|&h| (h as f64 - expect_per_slot).powi(2) / expect_per_slot).sum();
let chi2_z = (chi2 - (s as f64 - 1.0)) / (2.0 * (s as f64 - 1.0)).sqrt();
let hot = *st.slot_hist.iter().max().unwrap() as f64 / expect_per_slot;
let cold = *st.slot_hist.iter().min().unwrap() as f64 / expect_per_slot;
println!(
"{name} | {s} | {per_lane_hash} | {} | {} | {got_pct:.2} | {exp_pct:.2} | {chi2_z:.1} (hottest slot {hot:.2}x, coldest {cold:.2}x) | {} | {worst:.2} | {worst_delta:.2}",
st.events, st.hits, st.max_depth
);
// a regression band, not a uniformity claim: the rate sits between the uniform birthday rate and twice it
assert!(
got_pct >= 0.9 * exp_pct && got_pct <= 2.0 * exp_pct,
"{name}: re-hit rate {got_pct:.2}% against birthday {exp_pct:.2}%"
);
// depth histogram: the number of earlier RMWs a re-hit slot had seen in the unit
let shown: Vec<String> = st.depth.iter().take(st.max_depth + 1).enumerate().map(|(d, n)| format!("{d}:{n}")).collect();
println!(" depth histogram {}", shown.join(" "));
}
}
// ---------------------------------------------------------------------------------------------------------------
// 3. Hand-built edge programs against an independent hand model (question 3, CPU half)
// ---------------------------------------------------------------------------------------------------------------
fn ins(op: Op, dst: u8, src: u8) -> Instr {
Instr { op, dst, src, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 }
}
fn add_imm(dst: u8, src: u8, imm: u32) -> Instr {
Instr { op: Op::Add, dst, src, src2: 0, imm, imm2: imm, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 }
}
/// A hand-built program of class `c` named `name` (its seed is the name, so its fill words and init words are
/// its own). These bypass the generator and the acceptance rule, like `TESTS.md` section 2; `sub r, r` zeroes a
/// register as the Swift edge set does.
fn edge(name: &str, c: LoadClass, instrs: Vec<Instr>) -> Program {
let seed_string = format!("igneum-scratch-edge/{name}");
let seed_bytes = seed_string.as_bytes().to_vec();
let k = instrs.iter().filter(|i| i.op == Op::Scratch).count();
assert_eq!(k, c.scratch_slots(), "{name}: the class carries the program's scratch count");
Program {
seed: seed_words_from_bytes(&seed_bytes),
seed_string,
seed_bytes,
generator: GENERATOR_VERSION,
attempt: 0,
class: c,
era_bytes: None,
instrs,
}
}
/// The edge set for a scratch of `kb` KiB per warp. Each entry: (name, what it drives, program).
fn edge_programs(kb: u8) -> Vec<(String, &'static str, Program)> {
let m = LoadClass::scratch(1, kb).scratch_slot_mask();
let dsts = [2u8, 3, 4, 5, 6, 7, 0, 2, 3, 4, 5, 6, 7, 0, 2, 3];
let scr = |n: usize, src: u8| -> Vec<Instr> { (0..n).map(|i| ins(Op::Scratch, dsts[i], src)).collect() };
let mut v = Vec::new();
// every RMW of the hash to slot 0 through a zero register: 64 dependent RMWs on one slot per lane
let mut p = vec![ins(Op::Sub, 1, 1)];
p.extend(scr(8, 1));
v.push(("slot0".to_string(), "r1 = 0: every RMW to slot 0", edge(&format!("slot0/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot MASK through the in-range register MASK
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m)];
p.extend(scr(8, 1));
v.push(("slotmask".to_string(), "r1 = MASK: every RMW to the last slot", edge(&format!("slotmask/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot MASK through the out-of-range register 0xffffffff
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, 1), ins(Op::Sub, 1, 2)];
p.extend(scr(8, 1));
v.push(("ones".to_string(), "r1 = 0xffffffff: masked to the last slot", edge(&format!("ones/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot 0 through the out-of-range register MASK + 1
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m.wrapping_add(1))];
p.extend(scr(8, 1));
v.push(("maskplus1".to_string(), "r1 = MASK + 1: masked to slot 0", edge(&format!("maskplus1/k{kb}"), LoadClass::scratch(8, kb), p)));
// 16 RMWs per iteration on slot 0: 128 dependent RMWs on one slot per lane per hash
let mut p = vec![ins(Op::Sub, 1, 1)];
p.extend(scr(16, 1));
v.push(("sixteen".to_string(), "16 RMWs per iteration on slot 0", edge(&format!("sixteen/k{kb}"), LoadClass::scratch(16, kb), p)));
// one slot per lane from the init words: lanes with equal slots would show any cross-lane aliasing
// (r5 is the slot register and is never a destination here)
let p: Vec<Instr> = [0u8, 1, 2, 3, 4, 6, 7, 0].iter().map(|&d| ins(Op::Scratch, d, 5)).collect();
v.push(("lanevar".to_string(), "r5 never written: one init-dependent slot per lane", edge(&format!("lanevar/k{kb}"), LoadClass::scratch(8, kb), p)));
// alternating slot 0 and slot MASK inside one iteration
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, m)];
for (i, &d) in [3u8, 4, 5, 6, 7, 0, 3, 4].iter().enumerate() {
// r1 and r2 hold the two slots and are never destinations
p.push(ins(Op::Scratch, d, if i % 2 == 0 { 1 } else { 2 }));
}
v.push(("twoslots".to_string(), "slot 0 and slot MASK alternating", edge(&format!("twoslots/k{kb}"), LoadClass::scratch(8, kb), p)));
v
}
/// The hand model: a second, minimal interpreter for the ops the edge programs use (sub, add, scratch), with its
/// own slot store keyed by (lane, slot). `mutate` swaps the rewrite's words to show the comparison has teeth.
fn hand_model(p: &Program, base: u32, mutate: bool) -> [u64; 32] {
let seed = &p.seed;
let m = p.class.scratch_slot_mask();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
let nonce = base.wrapping_add(lane as u32);
for i in 0..8 {
let mut x = nonce ^ seed[i];
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
x = splitmix32(x);
r[i][lane] = x ^ seed[(i + 1) & 7];
}
}
let mut store: HashMap<(usize, u32), [u32; 3]> = HashMap::new();
for _ in 0..ITERATIONS {
let sel = r[0];
for ins in &p.instrs {
let (d, a) = (ins.dst as usize, ins.src as usize);
match ins.op {
Op::Sub => {
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_sub(r[a][lane]);
}
}
Op::Add => {
for lane in 0..LANES {
let c = if (sel[lane] >> ins.bit) & 1 != 0 { ins.imm2 } else { ins.imm };
r[d][lane] = r[d][lane].wrapping_add(r[a][lane]).wrapping_add(c);
}
}
Op::Scratch => {
for lane in 0..LANES {
let slot = r[a][lane] & m;
let w = *store.entry((lane, slot)).or_insert_with(|| {
let mut f = [0u32; 3];
for j in 0..3u32 {
// the fill, written out in full rather than through verify::scratch_fill
let n = base.wrapping_add(lane as u32);
f[j as usize] = splitmix32(
(n ^ seed[j as usize])
.wrapping_add(slot.wrapping_mul(0x9E37_79B1))
.wrapping_add((j + 1).wrapping_mul(0x85EB_CA77)),
);
}
f
});
let mut x = r[d][lane] ^ w[0];
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[1];
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[2];
r[d][lane] = x;
let out = if mutate {
[x.rotate_left(7) ^ w[2], x ^ w[1], x.wrapping_add(w[0])]
} else {
[x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])]
};
store.insert((lane, slot), out);
}
}
other => panic!("the hand model does not implement {other:?}"),
}
}
}
let mut out = [0u64; 32];
for lane in 0..LANES {
let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21);
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
out[lane] = ((hi as u64) << 32) | lo as u64;
}
out
}
/// The four unit bases of every edge vector: 0 and 32 (two consecutive units, the pair a one-warp persistent
/// launch runs on one arena), a unit straddling 2^31, and the unit that wraps past 2^32.
const EDGE_BASES: [u32; 4] = [0, 32, 0x7fff_fff0, 0xffff_ffe0];
#[test]
fn edge_programs_match_the_hand_model() {
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24);
let mut cases = 0;
for kb in [32u8, 128] {
for (name, what, p) in edge_programs(kb) {
let slots = p.class.scratch_slots_per_lane();
for base in EDGE_BASES {
let (res, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true);
let hand = hand_model(&p, base, false);
assert_eq!(res.hashes, hand, "{name} k{kb} base {base:#x}: interpreter against the hand model ({what})");
assert_ne!(res.hashes, hand_model(&p, base, true), "{name} k{kb}: the comparison has teeth");
// the slots the trace saw are the ones the program was built to drive
let slot_set: std::collections::BTreeSet<u32> = ev.iter().map(|e| e.slot).collect();
let m = (slots - 1) as u32;
match name.as_str() {
"slot0" | "maskplus1" | "sixteen" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![0]),
"slotmask" | "ones" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![m]),
"twoslots" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![0, m]),
"lanevar" => {
for e in &ev {
assert!(e.slot <= m);
}
}
_ => unreachable!(),
}
// the chain depth on the driven slot: every RMW after the first per lane is a re-hit
let per_lane = p.scratch_ops_per_hash();
let hits = ev.iter().filter(|e| e.hit).count();
let expected_hits = match name.as_str() {
"twoslots" => (per_lane - 2) * LANES,
_ => (per_lane - 1) * LANES,
};
assert_eq!(hits, expected_hits, "{name} k{kb}: re-hits");
cases += 1;
}
}
}
assert_eq!(cases, 2 * 7 * 4);
}
// ---------------------------------------------------------------------------------------------------------------
// 4. The static scratch check over every emitted kernel of every scr pack (question 4)
// ---------------------------------------------------------------------------------------------------------------
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum Dialect {
Metal,
Cuda,
OpenCl,
}
/// The static scratch check: every scratch read-modify-write in an emitted kernel has the one masked form the
/// emitter writes, the arena is the lane's own `slots x 4` words, the tag is `salt + unit`, and nothing else
/// touches the scratch. Like the dataset mask check of `TESTS.md` section 5 and `tests/packs.rs`, a text check:
/// the guarantee is that the emitter has one template and it masks.
pub fn scratch_text_check(text: &str, dialect: Dialect, k: usize, slots: usize, kernels: usize) -> Result<(), String> {
assert!(kernels >= 1);
// every count below is per hash kernel; an OpenCL bound file carries igneum_hash and igneum_hash_bound
let k = k * kernels;
assert!(slots.is_power_of_two() && slots >= 1);
let mask = (slots - 1) as u32;
let wpl = slots * 4;
let (u, load, store, ptr) = match dialect {
Dialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "device uint* arena"),
Dialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "uint32_t* arena"),
Dialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); }", "__global uint* arena"),
};
let count = |needle: &str| text.matches(needle).count();
let mut errs = Vec::new();
let mut expect = |what: &str, got: usize, want: usize| {
if got != want {
errs.push(format!("{what}: {got}, expected {want}"));
}
};
// k slot computations, each masked with exactly the class's mask and immediately followed by the one load form
expect("slot definitions `{ u s_ = r`", count(&format!("{{ {u} s_ = r")), k);
expect("masked slot followed by the load", count(&format!(" & {mask}u; {load}")), k);
expect("stores of the tagged slot", count(store), k);
expect("tag compares", count("(v_.x == tag)"), k);
expect("fill calls (three per RMW)", count("scr_fill(gbase, lane, s_, "), 3 * k);
// the arena: one definition with the class's words per lane, and 2k uses (one load, one store per RMW)
expect("arena definition", count(&format!("{ptr} = scratch + ((size_t)warp_ * 32u + lane) * {wpl}u;")), kernels);
expect("arena mentions (definition + load + store per RMW)", count("arena"), kernels + 2 * k);
expect("tag definition `tag = salt + g_`", count(&format!("{u} tag = salt + g_;")), kernels);
expect("direct scratch indexing", count("scratch["), 0);
expect("scratch pointer arithmetic outside the arena definition", count("scratch +"), kernels);
// no other mask value on a slot: every `s_ = r` line carries the class mask and nothing else carries ` & Nu; uint4 v_`
let any_mask_load = count(&format!("u; {load}"));
expect("loads preceded by some mask (must all be the class mask)", any_mask_load, k);
if errs.is_empty() {
Ok(())
} else {
Err(errs.join("; "))
}
}
fn packs_rw_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-readwidth")
}
fn scr_packs() -> Vec<String> {
let mut v: Vec<String> = std::fs::read_dir(packs_rw_dir())
.unwrap()
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
.filter(|n| n.starts_with("scr"))
.collect();
v.sort();
v
}
fn read_pack(pack: &str, file: &str) -> String {
let p = packs_rw_dir().join(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
/// Every scr pack regenerates from its program.json (seed bytes, class, day bytes, size) to the same six kernel
/// texts, byte for byte, and every one of those texts passes the static scratch check for the class's k and slot
/// count; the check fails on four deliberate breaks of a copy of the Metal text (mask dropped, mask changed, arena
/// stride changed, a stray scratch access) and on the OpenCL and CUDA twins of the first.
#[test]
fn scr_packs_regenerate_and_pass_the_static_scratch_check() {
let packs = scr_packs();
assert!(packs.len() >= 6, "the scr packs: {packs:?}");
let mut checked = 0;
let mut sample_metal = String::new();
let mut sample_cl = String::new();
let mut sample_cu = String::new();
let mut sample_k = 0;
let mut sample_slots = 0;
for pack in &packs {
let j: Value = serde_json::from_str(&read_pack(pack, "program.json")).unwrap();
let name = j["load_class"].as_str().unwrap();
let c = class(name);
assert_eq!(&format!("{name}"), pack, "pack directory named after its class");
let seed = j["seed"].as_str().unwrap();
let seed_bytes = igneum_pow::bind::unhex(j["seed_bytes"].as_str().unwrap()).unwrap();
let day_bytes = igneum_pow::bind::unhex(j["dataset"]["day_bytes"].as_str().unwrap()).unwrap();
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard");
let program = generate_from_seed_bytes_class(seed, &seed_bytes, c);
assert_eq!(program.class, c);
assert_eq!(program.program_id(), u64::from_str_radix(j["program_id"].as_str().unwrap().trim_start_matches("0x"), 16).unwrap());
let mut dataset = DatasetSource::from_key(seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2);
dataset.key_bytes = day_bytes;
let e = Epoch { program, dataset };
let p = &e.program;
let mp = e.dataset.memhard().map(|m| &m.params);
let k = c.scratch_slots();
let slots = c.scratch_slots_per_lane();
assert_eq!(p.scratch_ops_per_hash(), k * ITERATIONS);
for (file, text, dialect, kernels) in [
("program.metal", metal_program(p, log2, LoadSource::Stored), Dialect::Metal, 1),
("program_bound.metal", metal_program_bound(p, log2), Dialect::Metal, 1),
("kernel.cu", cuda_kernel(p, mp), Dialect::Cuda, 1),
("kernel_bound.cu", cuda_kernel_bound(p, mp), Dialect::Cuda, 1),
("kernel.cl", opencl_kernel(p, mp), Dialect::OpenCl, 1),
// the OpenCL bound file carries igneum_hash and igneum_hash_bound
("kernel_bound.cl", opencl_kernel_bound(p, mp), Dialect::OpenCl, 2),
] {
let on_disk = read_pack(pack, file);
assert_eq!(on_disk, text, "{pack}/{file}: the pack is the emitter's text");
// scr0 is the persistent control: an arena and a tag, no read-modify-write; the check holds with k = 0
scratch_text_check(&on_disk, dialect, k, slots, kernels).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"));
checked += 1;
}
// the vectors of the pack are the CPU's
let v: Value = serde_json::from_str(&read_pack(pack, "vectors.json")).unwrap();
for w in v["warps"].as_array().unwrap() {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let got = e.hash_warp(base);
for (lane, x) in w["expected"].as_array().unwrap().iter().enumerate() {
let want = u64::from_str_radix(x.as_str().unwrap().trim_start_matches("0x"), 16).unwrap();
assert_eq!(got[lane], want, "{pack}: base {base} lane {lane}");
}
}
if k == 4 && slots == 64 {
sample_metal = read_pack(pack, "program.metal");
sample_cl = read_pack(pack, "kernel.cl");
sample_cu = read_pack(pack, "kernel.cu");
sample_k = k;
sample_slots = slots;
}
}
assert_eq!(checked, packs.len() * 6);
println!("static scratch check: {checked} kernels over {} scr packs", packs.len());
// The deliberate breaks (the watcher rule of CLAUDE.md: a check is trusted once it fails on a known-broken
// case). Each must be caught; the message names what.
assert!(sample_k == 4 && sample_slots == 64, "scr4k32 is in the pack set");
let mask = format!(" & {}u; uint4 v_", sample_slots - 1);
let broken_mask = sample_metal.replacen(&mask, "; uint4 v_", 1);
assert_ne!(broken_mask, sample_metal);
let e = scratch_text_check(&broken_mask, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 1 (one mask dropped, Metal): {e}");
let wrong_mask = sample_metal.replace(" & 63u;", " & 127u;");
let e = scratch_text_check(&wrong_mask, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 0, expected 4"), "{e}");
println!("break 2 (mask 63 -> 127 on every RMW, Metal): {e}");
let wrong_stride = sample_metal.replace("* 256u;", "* 128u;");
let e = scratch_text_check(&wrong_stride, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("arena definition: 0, expected 1"), "{e}");
println!("break 3 (arena stride 256 -> 128 words, Metal): {e}");
let stray = format!("{sample_metal}\n// stray\n// arena[0] = 0u; scratch[1] = 1u;\n");
let e = scratch_text_check(&stray, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("arena mentions") && e.contains("direct scratch indexing: 1, expected 0"), "{e}");
println!("break 4 (a stray arena and scratch access, Metal): {e}");
let e = scratch_text_check(&sample_cl.replacen(" & 63u; uint4 v_ = vload4", "; uint4 v_ = vload4", 1), Dialect::OpenCl, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 5 (one mask dropped, OpenCL): {e}");
let e = scratch_text_check(&sample_cu.replacen(" & 63u; uint4 v_ = *(const uint4*)", "; uint4 v_ = *(const uint4*)", 1), Dialect::Cuda, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 6 (one mask dropped, CUDA): {e}");
// and the unbroken texts pass under the same calls
scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 1).unwrap();
scratch_text_check(&sample_cl, Dialect::OpenCl, 4, 64, 1).unwrap();
scratch_text_check(&sample_cu, Dialect::Cuda, 4, 64, 1).unwrap();
// a wrong slot count, RMW count or kernel count against a right text fails too (the check is tied to the class)
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 256, 1).is_err());
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 3, 64, 1).is_err());
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 2).is_err());
}
// ---------------------------------------------------------------------------------------------------------------
// 5. The fuzz: 200 generated scratch programs, contract on every instruction, 4 units each across the 32-bit
// range including the wrap; with IGNEUM_SCRATCH_PACKS_OUT the packs for the Metal runs (question 3, 4)
// ---------------------------------------------------------------------------------------------------------------
/// Write a pack whose vectors.json carries `bases` (any number of units) instead of the three standard bases.
fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) -> Vec<[u64; 32]> {
let mut pack = export_pack(e, day, source);
let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect();
let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true);
for f in pack.files.iter_mut() {
if f.0 == "vectors.json" {
f.1 = vj.clone();
}
}
pack.write_to(dir).unwrap();
outs
}
fn contract(p: &Program) {
assert_eq!(p.instrs.len(), INSTR_COUNT);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count() + p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), 16);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), p.class.scratch_slots());
assert!(p.instrs[0].op != Op::Load && p.instrs[0].op != Op::Scratch, "instruction 0 is never a memory op");
for (k, i) in p.instrs.iter().enumerate() {
assert!(i.src != i.dst, "#{k}: src == dst");
assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot);
assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask);
assert!(i.dst < 8 && i.src < 8 && i.src2 < 8);
assert_eq!(i.width, 1, "#{k}: a scratch class reads one-word loads");
}
assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program");
}
#[test]
fn fuzz_scr_programs_cpu() {
let n: usize = std::env::var("IGNEUM_SCRATCH_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200);
let out = std::env::var("IGNEUM_SCRATCH_PACKS_OUT").ok().map(PathBuf::from);
let mut rng = SplitMix64::new(0x6967_6e65_756d_2d73); // "igneum-s"
let day = "2026-10-03";
let closed = DatasetSource::new(day, DatasetMode::ClosedForm, 28);
// memory-hard sources per size, built once each (the cache fill is 0.2 s); only when packs are written
let mut mh: HashMap<u32, DatasetSource> = HashMap::new();
let mut manifest = String::from("pack\tclass\tlog2\tprogram_id\tscratch_ops_per_hash\tbases\n");
let mut per_class: HashMap<String, usize> = HashMap::new();
let mut units = 0usize;
let mut wraps = 0usize;
if let Some(dir) = &out {
std::fs::create_dir_all(dir).unwrap();
// the edge packs first: 64 MiB datasets (no dataset load in them), the four edge bases
for kb in [32u8, 128] {
for (name, _what, p) in edge_programs(kb) {
let log2 = 24;
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2));
let e = Epoch { program: p, dataset: ds };
let pack_name = format!("edge-{name}-k{kb}");
write_pack_with_bases(&dir.join(&pack_name), &e, day, &EDGE_BASES, "igneum-pow tests/scratch.rs edge");
manifest.push_str(&format!(
"{pack_name}\t{}\t{log2}\t{:016x}\t{}\t{}\n",
e.program.class.name(),
e.program.program_id(),
e.program.scratch_ops_per_hash(),
EDGE_BASES.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
));
mh.insert(log2, e.dataset);
}
}
}
for i in 0..n {
let name = CLASSES[rng.below(CLASSES.len() as u64) as usize];
let c = class(name);
let seed = format!("igneum-scratch-fuzz/{i}");
let p = generate_class(&seed, c);
contract(&p);
*per_class.entry(name.to_string()).or_insert(0) += 1;
// four bases: one inside a 256-nonce batch (in-batch check on the GPU), one straddling 2^31, one in
// the last 256 nonces (the unit wraps past 2^32 or ends on it), one uniform
let b0 = (rng.below(8) as u32) * 32;
let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32);
let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32);
let b3 = (rng.next() as u32) & !31;
let bases = [b0, b1, b2, b3];
// an aligned unit never straddles 2^32 (spec 1.9); the top unit ends on 0xffffffff and the persistent
// kernel's unit sequence wraps inside a launch, which the Metal run checks with packbench --batch-base
wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count();
// the CPU: the interpreter is deterministic and every scratch event is inside the lane's slots
for &b in &bases {
let (r1, ev) = interpret_warp_scratch(&p, &p.seed, b, &closed, true);
let r2 = interpret_warp_scratch(&p, &p.seed, b, &closed, false).0;
assert_eq!(r1.hashes, r2.hashes);
assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES);
assert!(ev.iter().all(|e: &ScratchEvent| e.slot < c.scratch_slots_per_lane() as u32));
units += 1;
}
if let Some(dir) = &out {
let log2 = [24u32, 26, 28][rng.below(3) as usize];
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2));
let e = Epoch { program: p, dataset: ds };
let pack_name = format!("fuzz-{i:03}-{name}-l{log2}");
write_pack_with_bases(&dir.join(&pack_name), &e, day, &bases, "igneum-pow tests/scratch.rs fuzz");
manifest.push_str(&format!(
"{pack_name}\t{name}\t{log2}\t{:016x}\t{}\t{}\n",
e.program.program_id(),
e.program.scratch_ops_per_hash(),
bases.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
));
mh.insert(log2, e.dataset);
} else {
let _ = rng.below(3);
}
}
let mut classes: Vec<_> = per_class.iter().collect();
classes.sort();
println!("fuzz: {n} programs, {units} units on the CPU, {wraps} units in the top 256 nonces, classes {classes:?}");
assert_eq!(units, 4 * n);
assert_eq!(wraps, n, "every program has a unit in the top 256 nonces");
if let Some(dir) = &out {
std::fs::write(dir.join("manifest.tsv"), manifest).unwrap();
println!("packs written to {}", dir.display());
}
}
/// The fold and rewrite, restated: a slot after `d` dependent RMWs holds 96 bits that are a function of the fill
/// (3 words, a pure function of nonce, slot and seed) and the `d` fold values; a chip that keeps the `d` fold
/// values (32 bits each) instead of the 96-bit slot recomputes the slot in `d` rewrites. This test pins the
/// arithmetic the analysis uses (question 2): the replay from the fold values reproduces the slot.
#[test]
fn slot_is_replayable_from_its_fold_values() {
let seed = seed_words_from_bytes(b"igneum-genesis");
let (base, lane, slot) = (0x1234_5600u32, 5u32, 17u32);
let fill = [scratch_fill(&seed, base, lane, slot, 0), scratch_fill(&seed, base, lane, slot, 1), scratch_fill(&seed, base, lane, slot, 2)];
let mut rng = SplitMix64::new(99);
let dsts: Vec<u32> = (0..64).map(|_| rng.next() as u32).collect();
// the honest sequence: read, fold, rewrite, 64 times
let mut w = fill;
let mut xs = Vec::new();
for &d in &dsts {
let x = fold_words(d, &w);
xs.push(x);
w = scratch_rewrite(x, &w);
}
// the replay: from the fill and the stored fold values alone
let mut w2 = fill;
for &x in &xs {
w2 = scratch_rewrite(x, &w2);
}
assert_eq!(w, w2);
// and nothing shorter: the fold value at step d depends on the slot content at step d, which depends on
// every earlier fold value (drop one and the chain diverges)
let mut w3 = fill;
for (i, &x) in xs.iter().enumerate() {
if i != 10 {
w3 = scratch_rewrite(x, &w3);
}
}
assert_ne!(w, w3);
}