//! Soundness tests of layer 3 of `docs/plans/counter-asic-2.md`: the per-warp scratch with read-modify-writes //! (variant 5 of the read-width experiment, `LoadClass::scratch(k, kb)`). Analysis and results: //! `docs/analysis/scratch-soundness.md`. Every test is parametric over the class's slot count //! (`scratch_slots_per_lane()`), so the 32 and 128 KiB geometries and any later one run the same checks. //! //! What runs under plain `cargo test`: //! 1. `rewrite_is_a_bijection_of_the_fold_value`, `fill_is_a_bijection_of_the_nonce`: the written words as //! functions (question 1). //! 2. `written_words_unbiased_and_rehit_rates`: bit bias of every written word over 2^11 units x 3 seeds per class //! (the TESTS.md section 3 shape), and the measured slot re-hit rate against the birthday formula (question 2). //! 3. `edge_programs_match_the_hand_model`: hand-built programs that drive every read-modify-write of a hash to //! slot 0, slot MASK, through out-of-range registers, to one slot per lane, alternating two slots, and 16 //! read-modify-writes per iteration on one slot; the interpreter against an independent hand model, and the //! hand model shown to have teeth (question 3, CPU half). //! 4. `scr_packs_regenerate_and_pass_the_static_scratch_check`: every emitted kernel of every scr pack under //! `proto-cuda/packs-readwidth` regenerates from its program.json and passes the static scratch-mask check; //! the check is shown to fail on four deliberate breaks (question 4). //! 5. `fuzz_scr_programs_cpu`: 200 generated scratch programs over the six classes, generator contract on every //! instruction, 4 units each at base nonces across the 32-bit range including the wrap; with //! `IGNEUM_SCRATCH_PACKS_OUT=` it also writes the packs (and the edge packs) for the Metal runs of //! `proto-metal/packbench` (question 3 GPU half, question 4, `TESTS.md` section 9 shape). use igneum_pow::emit::{ cuda_kernel, cuda_kernel_bound, export_pack, metal_program, metal_program_bound, opencl_kernel, opencl_kernel_bound, vectors_json, LoadSource, }; use igneum_pow::generator::{ generate_class, generate_from_seed_bytes_class, Instr, LoadClass, Op, Program, GENERATOR_VERSION, INSTR_COUNT, ITERATIONS, LANES, }; use igneum_pow::seed::{seed_words_from_bytes, SplitMix64}; use igneum_pow::verify::{ fold_words, interpret_warp_scratch, scratch_fill, scratch_rewrite, splitmix32, DatasetMode, DatasetSource, Epoch, ScratchEvent, FOLD_MUL, FOLD_ROT, }; use serde_json::Value; use std::collections::HashMap; use std::path::PathBuf; /// The classes under study: the two capped geometries (32 and 128 KiB per warp: 64 and 256 slots per lane) at the /// RMW shares the readwidth branch measures. const CLASSES: [&str; 6] = ["scr2k32", "scr4k32", "scr8k32", "scr2k128", "scr4k128", "scr8k128"]; fn class(name: &str) -> LoadClass { LoadClass::parse(name).unwrap_or_else(|| panic!("class {name}")) } // --------------------------------------------------------------------------------------------------------------- // 1. The written words as functions (question 1) // --------------------------------------------------------------------------------------------------------------- /// For a fixed slot content `w`, each of the three rewritten words is a bijection of the fold value `x` /// (`x ^ w1`, `rotl(x, 7) ^ w2`, `x + w0`), so the rewrite is injective in `x` and a uniform `x` gives a uniform /// word in every position. Checked over 2^16 consecutive `x` for 16 random `w`. #[test] fn rewrite_is_a_bijection_of_the_fold_value() { let mut rng = SplitMix64::new(0x7363_7261_7463_6801); for _ in 0..16 { let w = [rng.next() as u32, rng.next() as u32, rng.next() as u32]; let x0 = rng.next() as u32; let mut seen = [vec![false; 1 << 16], vec![false; 1 << 16], vec![false; 1 << 16]]; for i in 0..(1u32 << 16) { let x = x0.wrapping_add(i); let out = scratch_rewrite(x, &w); for j in 0..3 { // a bijection of x maps 2^16 consecutive x to 2^16 distinct words; the low 16 bits alone are // distinct for the xor words (x ^ c) and for the add word (x + c), since both act on the low 16 // bits as bijections of the low 16 bits of x; the rotl word is checked on its rotated-back bits let key = if j == 1 { out[j].rotate_right(7) & 0xffff } else { out[j] & 0xffff }; assert!(!seen[j][key as usize], "word {j} repeats inside 2^16 consecutive x"); seen[j][key as usize] = true; } } } // The rewrite inverts: from the old content and any ONE written word the fold value is recovered, so a // rewritten slot carries exactly 32 bits of new state (the point of question 2's arithmetic). let w = [0x1234_5678, 0x9abc_def0, 0x0fed_cba9]; let x = 0xdead_beef; let out = scratch_rewrite(x, &w); assert_eq!(out[0] ^ w[1], x); assert_eq!((out[1] ^ w[2]).rotate_right(7), x); assert_eq!(out[2].wrapping_sub(w[0]), x); } /// For a fixed (seed, slot, j) the fill is a bijection of the lane nonce: `splitmix32` is a bijection of its /// 32-bit input and the input `((base + lane) ^ s) + c` is a bijection of `base + lane`. Over 2^16 consecutive /// nonces no fill word repeats, for 8 slots x 3 words. #[test] fn fill_is_a_bijection_of_the_nonce() { let seed = seed_words_from_bytes(b"igneum-genesis"); for slot in [0u32, 1, 63, 64, 255, 1023, 2047] { for j in 0..3u32 { let mut words: Vec = (0..(1u32 << 16)).map(|n| scratch_fill(&seed, n, 0, slot, j)).collect(); words.sort_unstable(); words.dedup(); assert_eq!(words.len(), 1 << 16, "slot {slot} word {j}: fill words of 2^16 consecutive nonces are distinct"); } } // base + lane is the lane nonce: the fill of lane l at base b is the fill of lane 0 at base b + l assert_eq!(scratch_fill(&seed, 0x1000, 7, 5, 2), scratch_fill(&seed, 0x1007, 0, 5, 2)); // and it wraps with the nonce: base 0xffffffe0, lane 31 is nonce 0xffffffff; lane 32 would be nonce 0 assert_eq!(scratch_fill(&seed, 0xffff_ffe0, 32, 5, 2), scratch_fill(&seed, 0, 0, 5, 2)); // the three word positions of one slot and nonce are three different permutation outputs let f: Vec = (0..3).map(|j| scratch_fill(&seed, 12345, 7, 17, j)).collect(); assert!(f[0] != f[1] && f[1] != f[2] && f[0] != f[2]); } // --------------------------------------------------------------------------------------------------------------- // 2. Uniformity of the written words and the slot re-hit rate (questions 1 and 2) // --------------------------------------------------------------------------------------------------------------- /// Birthday arithmetic: the expected number of distinct slots after `n` uniform draws from `s` slots. fn expected_distinct(s: usize, n: usize) -> f64 { let s = s as f64; s * (1.0 - (1.0 - 1.0 / s).powi(n as i32)) } struct ClassStats { units: usize, events: usize, hits: usize, /// ones count per bit of the written words, 3 x 32 ones: [[u64; 32]; 3], /// ones count per bit of written XOR read (the change the rewrite makes to the slot) delta_ones: [[u64; 32]; 3], /// re-hit depth histogram: how many earlier RMWs the slot had seen in this unit (0 = first touch) depth: Vec, max_depth: usize, /// how often each slot index was addressed (the slot comes from a register's low bits) slot_hist: Vec, } fn class_stats(name: &str, seeds: &[&str], units_per_seed: usize) -> ClassStats { let c = class(name); let mut st = ClassStats { units: 0, events: 0, hits: 0, ones: [[0; 32]; 3], delta_ones: [[0; 32]; 3], depth: vec![0; 256], max_depth: 0, slot_hist: vec![0; c.scratch_slots_per_lane()], }; let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28); for seed in seeds { let p = generate_class(seed, c); assert_eq!(p.scratch_ops_per_hash(), c.scratch_slots() * ITERATIONS); for u in 0..units_per_seed { let base = (u as u32).wrapping_mul(32).wrapping_add(0x4000_0000); let (_, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true); assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES); let mut count: HashMap<(u8, u32), usize> = HashMap::new(); for e in &ev { assert!(e.slot < c.scratch_slots_per_lane() as u32, "slot inside the lane's scratch"); let d = count.entry((e.lane, e.slot)).or_insert(0); assert_eq!(e.hit, *d > 0, "hit flag agrees with the unit's own history"); assert_eq!(e.written, scratch_rewrite(e.x, &e.read)); if !e.hit { let fill = [ scratch_fill(&p.seed, base, e.lane as u32, e.slot, 0), scratch_fill(&p.seed, base, e.lane as u32, e.slot, 1), scratch_fill(&p.seed, base, e.lane as u32, e.slot, 2), ]; assert_eq!(e.read, fill, "a first touch reads the fill"); } st.depth[(*d).min(255)] += 1; st.max_depth = st.max_depth.max(*d); st.slot_hist[e.slot as usize] += 1; *d += 1; st.events += 1; st.hits += e.hit as usize; for j in 0..3 { for b in 0..32 { st.ones[j][b] += ((e.written[j] >> b) & 1) as u64; st.delta_ones[j][b] += (((e.written[j] ^ e.read[j]) >> b) & 1) as u64; } } } st.units += 1; } } st } /// Bit bias of every written word (and of the change each rewrite makes) within 6 sigma of a fair coin, over /// 3 seeds x 2^11 units per class (131,072 hashes per seed set); the slot re-hit rate against the birthday /// formula within 3 percent relative. The table printed here is the one in the analysis. #[test] fn written_words_unbiased_and_rehit_rates() { let seeds = ["igneum-genesis", "igneum-genesis/stats1", "igneum-genesis/stats2"]; let units = 1usize << 11; println!("class | slots/lane | RMW/hash | events | re-hits | re-hit % | birthday % | slot chi2 z (spread) | max depth | max bias sigma | max delta bias sigma"); for name in CLASSES { let c = class(name); let st = class_stats(name, &seeds, units); let n = st.events as f64; let sigma = (n / 4.0).sqrt(); let mut worst = 0.0f64; let mut worst_delta = 0.0f64; for j in 0..3 { for b in 0..32 { let z = (st.ones[j][b] as f64 - n / 2.0).abs() / sigma; let zd = (st.delta_ones[j][b] as f64 - n / 2.0).abs() / sigma; assert!(z <= 6.0, "{name}: written word {j} bit {b} biased: {z:.2} sigma"); assert!(zd <= 6.0, "{name}: rewrite delta word {j} bit {b} biased: {zd:.2} sigma"); worst = worst.max(z); worst_delta = worst_delta.max(zd); } } let per_lane_hash = c.scratch_slots() * ITERATIONS; let s = c.scratch_slots_per_lane(); let exp_hits = per_lane_hash as f64 - expected_distinct(s, per_lane_hash); let exp_pct = 100.0 * exp_hits / per_lane_hash as f64; let got_pct = 100.0 * st.hits as f64 / st.events as f64; // chi-square of the slot histogram against uniform (df = s - 1): the slot is a register's low bits, and // the measured re-hit rate runs above the uniform birthday rate (the finding of the analysis, question 2) let expect_per_slot = n / s as f64; let chi2: f64 = st.slot_hist.iter().map(|&h| (h as f64 - expect_per_slot).powi(2) / expect_per_slot).sum(); let chi2_z = (chi2 - (s as f64 - 1.0)) / (2.0 * (s as f64 - 1.0)).sqrt(); let hot = *st.slot_hist.iter().max().unwrap() as f64 / expect_per_slot; let cold = *st.slot_hist.iter().min().unwrap() as f64 / expect_per_slot; println!( "{name} | {s} | {per_lane_hash} | {} | {} | {got_pct:.2} | {exp_pct:.2} | {chi2_z:.1} (hottest slot {hot:.2}x, coldest {cold:.2}x) | {} | {worst:.2} | {worst_delta:.2}", st.events, st.hits, st.max_depth ); // a regression band, not a uniformity claim: the rate sits between the uniform birthday rate and twice it assert!( got_pct >= 0.9 * exp_pct && got_pct <= 2.0 * exp_pct, "{name}: re-hit rate {got_pct:.2}% against birthday {exp_pct:.2}%" ); // depth histogram: the number of earlier RMWs a re-hit slot had seen in the unit let shown: Vec = st.depth.iter().take(st.max_depth + 1).enumerate().map(|(d, n)| format!("{d}:{n}")).collect(); println!(" depth histogram {}", shown.join(" ")); } } // --------------------------------------------------------------------------------------------------------------- // 3. Hand-built edge programs against an independent hand model (question 3, CPU half) // --------------------------------------------------------------------------------------------------------------- fn ins(op: Op, dst: u8, src: u8) -> Instr { Instr { op, dst, src, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 } } fn add_imm(dst: u8, src: u8, imm: u32) -> Instr { Instr { op: Op::Add, dst, src, src2: 0, imm, imm2: imm, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 } } /// A hand-built program of class `c` named `name` (its seed is the name, so its fill words and init words are /// its own). These bypass the generator and the acceptance rule, like `TESTS.md` section 2; `sub r, r` zeroes a /// register as the Swift edge set does. fn edge(name: &str, c: LoadClass, instrs: Vec) -> Program { let seed_string = format!("igneum-scratch-edge/{name}"); let seed_bytes = seed_string.as_bytes().to_vec(); let k = instrs.iter().filter(|i| i.op == Op::Scratch).count(); assert_eq!(k, c.scratch_slots(), "{name}: the class carries the program's scratch count"); Program { seed: seed_words_from_bytes(&seed_bytes), seed_string, seed_bytes, generator: GENERATOR_VERSION, attempt: 0, class: c, era_bytes: None, instrs, shadow: Vec::new(), } } /// The edge set for a scratch of `kb` KiB per warp. Each entry: (name, what it drives, program). fn edge_programs(kb: u8) -> Vec<(String, &'static str, Program)> { let m = LoadClass::scratch(1, kb).scratch_slot_mask(); let dsts = [2u8, 3, 4, 5, 6, 7, 0, 2, 3, 4, 5, 6, 7, 0, 2, 3]; let scr = |n: usize, src: u8| -> Vec { (0..n).map(|i| ins(Op::Scratch, dsts[i], src)).collect() }; let mut v = Vec::new(); // every RMW of the hash to slot 0 through a zero register: 64 dependent RMWs on one slot per lane let mut p = vec![ins(Op::Sub, 1, 1)]; p.extend(scr(8, 1)); v.push(("slot0".to_string(), "r1 = 0: every RMW to slot 0", edge(&format!("slot0/k{kb}"), LoadClass::scratch(8, kb), p))); // slot MASK through the in-range register MASK let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m)]; p.extend(scr(8, 1)); v.push(("slotmask".to_string(), "r1 = MASK: every RMW to the last slot", edge(&format!("slotmask/k{kb}"), LoadClass::scratch(8, kb), p))); // slot MASK through the out-of-range register 0xffffffff let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, 1), ins(Op::Sub, 1, 2)]; p.extend(scr(8, 1)); v.push(("ones".to_string(), "r1 = 0xffffffff: masked to the last slot", edge(&format!("ones/k{kb}"), LoadClass::scratch(8, kb), p))); // slot 0 through the out-of-range register MASK + 1 let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m.wrapping_add(1))]; p.extend(scr(8, 1)); v.push(("maskplus1".to_string(), "r1 = MASK + 1: masked to slot 0", edge(&format!("maskplus1/k{kb}"), LoadClass::scratch(8, kb), p))); // 16 RMWs per iteration on slot 0: 128 dependent RMWs on one slot per lane per hash let mut p = vec![ins(Op::Sub, 1, 1)]; p.extend(scr(16, 1)); v.push(("sixteen".to_string(), "16 RMWs per iteration on slot 0", edge(&format!("sixteen/k{kb}"), LoadClass::scratch(16, kb), p))); // one slot per lane from the init words: lanes with equal slots would show any cross-lane aliasing // (r5 is the slot register and is never a destination here) let p: Vec = [0u8, 1, 2, 3, 4, 6, 7, 0].iter().map(|&d| ins(Op::Scratch, d, 5)).collect(); v.push(("lanevar".to_string(), "r5 never written: one init-dependent slot per lane", edge(&format!("lanevar/k{kb}"), LoadClass::scratch(8, kb), p))); // alternating slot 0 and slot MASK inside one iteration let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, m)]; for (i, &d) in [3u8, 4, 5, 6, 7, 0, 3, 4].iter().enumerate() { // r1 and r2 hold the two slots and are never destinations p.push(ins(Op::Scratch, d, if i % 2 == 0 { 1 } else { 2 })); } v.push(("twoslots".to_string(), "slot 0 and slot MASK alternating", edge(&format!("twoslots/k{kb}"), LoadClass::scratch(8, kb), p))); v } /// The hand model: a second, minimal interpreter for the ops the edge programs use (sub, add, scratch), with its /// own slot store keyed by (lane, slot). `mutate` swaps the rewrite's words to show the comparison has teeth. fn hand_model(p: &Program, base: u32, mutate: bool) -> [u64; 32] { let seed = &p.seed; let m = p.class.scratch_slot_mask(); let mut r = [[0u32; LANES]; 8]; for lane in 0..LANES { let nonce = base.wrapping_add(lane as u32); for i in 0..8 { let mut x = nonce ^ seed[i]; x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1)); x = splitmix32(x); r[i][lane] = x ^ seed[(i + 1) & 7]; } } let mut store: HashMap<(usize, u32), [u32; 3]> = HashMap::new(); for _ in 0..ITERATIONS { let sel = r[0]; for ins in &p.instrs { let (d, a) = (ins.dst as usize, ins.src as usize); match ins.op { Op::Sub => { for lane in 0..LANES { r[d][lane] = r[d][lane].wrapping_sub(r[a][lane]); } } Op::Add => { for lane in 0..LANES { let c = if (sel[lane] >> ins.bit) & 1 != 0 { ins.imm2 } else { ins.imm }; r[d][lane] = r[d][lane].wrapping_add(r[a][lane]).wrapping_add(c); } } Op::Scratch => { for lane in 0..LANES { let slot = r[a][lane] & m; let w = *store.entry((lane, slot)).or_insert_with(|| { let mut f = [0u32; 3]; for j in 0..3u32 { // the fill, written out in full rather than through verify::scratch_fill let n = base.wrapping_add(lane as u32); f[j as usize] = splitmix32( (n ^ seed[j as usize]) .wrapping_add(slot.wrapping_mul(0x9E37_79B1)) .wrapping_add((j + 1).wrapping_mul(0x85EB_CA77)), ); } f }); let mut x = r[d][lane] ^ w[0]; x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[1]; x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[2]; r[d][lane] = x; let out = if mutate { [x.rotate_left(7) ^ w[2], x ^ w[1], x.wrapping_add(w[0])] } else { [x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])] }; store.insert((lane, slot), out); } } other => panic!("the hand model does not implement {other:?}"), } } } let mut out = [0u64; 32]; for lane in 0..LANES { let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21); let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27); out[lane] = ((hi as u64) << 32) | lo as u64; } out } /// The four unit bases of every edge vector: 0 and 32 (two consecutive units, the pair a one-warp persistent /// launch runs on one arena), a unit straddling 2^31, and the unit that wraps past 2^32. const EDGE_BASES: [u32; 4] = [0, 32, 0x7fff_fff0, 0xffff_ffe0]; #[test] fn edge_programs_match_the_hand_model() { let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24); let mut cases = 0; for kb in [32u8, 128] { for (name, what, p) in edge_programs(kb) { let slots = p.class.scratch_slots_per_lane(); for base in EDGE_BASES { let (res, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true); let hand = hand_model(&p, base, false); assert_eq!(res.hashes, hand, "{name} k{kb} base {base:#x}: interpreter against the hand model ({what})"); assert_ne!(res.hashes, hand_model(&p, base, true), "{name} k{kb}: the comparison has teeth"); // the slots the trace saw are the ones the program was built to drive let slot_set: std::collections::BTreeSet = ev.iter().map(|e| e.slot).collect(); let m = (slots - 1) as u32; match name.as_str() { "slot0" | "maskplus1" | "sixteen" => assert_eq!(slot_set.into_iter().collect::>(), vec![0]), "slotmask" | "ones" => assert_eq!(slot_set.into_iter().collect::>(), vec![m]), "twoslots" => assert_eq!(slot_set.into_iter().collect::>(), vec![0, m]), "lanevar" => { for e in &ev { assert!(e.slot <= m); } } _ => unreachable!(), } // the chain depth on the driven slot: every RMW after the first per lane is a re-hit let per_lane = p.scratch_ops_per_hash(); let hits = ev.iter().filter(|e| e.hit).count(); let expected_hits = match name.as_str() { "twoslots" => (per_lane - 2) * LANES, _ => (per_lane - 1) * LANES, }; assert_eq!(hits, expected_hits, "{name} k{kb}: re-hits"); cases += 1; } } } assert_eq!(cases, 2 * 7 * 4); } // --------------------------------------------------------------------------------------------------------------- // 4. The static scratch check over every emitted kernel of every scr pack (question 4) // --------------------------------------------------------------------------------------------------------------- #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum Dialect { Metal, Cuda, OpenCl, } /// The static scratch check: every scratch read-modify-write in an emitted kernel has the one masked form the /// emitter writes, the arena is the lane's own `slots x 4` words, the tag is `salt + unit`, and nothing else /// touches the scratch. Like the dataset mask check of `TESTS.md` section 5 and `tests/packs.rs`, a text check: /// the guarantee is that the emitter has one template and it masks. pub fn scratch_text_check(text: &str, dialect: Dialect, k: usize, slots: usize, kernels: usize) -> Result<(), String> { assert!(kernels >= 1); // every count below is per hash kernel; an OpenCL bound file carries igneum_hash and igneum_hash_bound let k = k * kernels; assert!(slots.is_power_of_two() && slots >= 1); let mask = (slots - 1) as u32; let wpl = slots * 4; let (u, load, store, ptr) = match dialect { Dialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "device uint* arena"), Dialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "uint32_t* arena"), Dialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); }", "__global uint* arena"), }; let count = |needle: &str| text.matches(needle).count(); let mut errs = Vec::new(); let mut expect = |what: &str, got: usize, want: usize| { if got != want { errs.push(format!("{what}: {got}, expected {want}")); } }; // k slot computations, each masked with exactly the class's mask and immediately followed by the one load form expect("slot definitions `{ u s_ = r`", count(&format!("{{ {u} s_ = r")), k); expect("masked slot followed by the load", count(&format!(" & {mask}u; {load}")), k); expect("stores of the tagged slot", count(store), k); expect("tag compares", count("(v_.x == tag)"), k); expect("fill calls (three per RMW)", count("scr_fill(gbase, lane, s_, "), 3 * k); // the arena: one definition with the class's words per lane, and 2k uses (one load, one store per RMW) expect("arena definition", count(&format!("{ptr} = scratch + ((size_t)warp_ * 32u + lane) * {wpl}u;")), kernels); expect("arena mentions (definition + load + store per RMW)", count("arena"), kernels + 2 * k); expect("tag definition `tag = salt + g_`", count(&format!("{u} tag = salt + g_;")), kernels); expect("direct scratch indexing", count("scratch["), 0); expect("scratch pointer arithmetic outside the arena definition", count("scratch +"), kernels); // no other mask value on a slot: every `s_ = r` line carries the class mask and nothing else carries ` & Nu; uint4 v_` let any_mask_load = count(&format!("u; {load}")); expect("loads preceded by some mask (must all be the class mask)", any_mask_load, k); if errs.is_empty() { Ok(()) } else { Err(errs.join("; ")) } } fn packs_rw_dir() -> PathBuf { PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-readwidth") } fn scr_packs() -> Vec { let mut v: Vec = std::fs::read_dir(packs_rw_dir()) .unwrap() .map(|d| d.unwrap().file_name().to_string_lossy().to_string()) .filter(|n| n.starts_with("scr")) .collect(); v.sort(); v } fn read_pack(pack: &str, file: &str) -> String { let p = packs_rw_dir().join(pack).join(file); std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display())) } /// Every scr pack regenerates from its program.json (seed bytes, class, day bytes, size) to the same six kernel /// texts, byte for byte, and every one of those texts passes the static scratch check for the class's k and slot /// count; the check fails on four deliberate breaks of a copy of the Metal text (mask dropped, mask changed, arena /// stride changed, a stray scratch access) and on the OpenCL and CUDA twins of the first. #[test] fn scr_packs_regenerate_and_pass_the_static_scratch_check() { let packs = scr_packs(); assert!(packs.len() >= 6, "the scr packs: {packs:?}"); let mut checked = 0; let mut sample_metal = String::new(); let mut sample_cl = String::new(); let mut sample_cu = String::new(); let mut sample_k = 0; let mut sample_slots = 0; for pack in &packs { let j: Value = serde_json::from_str(&read_pack(pack, "program.json")).unwrap(); let name = j["load_class"].as_str().unwrap(); let c = class(name); assert_eq!(&format!("{name}"), pack, "pack directory named after its class"); let seed = j["seed"].as_str().unwrap(); let seed_bytes = igneum_pow::bind::unhex(j["seed_bytes"].as_str().unwrap()).unwrap(); let day_bytes = igneum_pow::bind::unhex(j["dataset"]["day_bytes"].as_str().unwrap()).unwrap(); let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard"); let program = generate_from_seed_bytes_class(seed, &seed_bytes, c); assert_eq!(program.class, c); assert_eq!(program.program_id(), u64::from_str_radix(j["program_id"].as_str().unwrap().trim_start_matches("0x"), 16).unwrap()); let mut dataset = DatasetSource::from_key(seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2); dataset.key_bytes = day_bytes; let e = Epoch { program, dataset }; let p = &e.program; let mp = e.dataset.memhard().map(|m| &m.params); let k = c.scratch_slots(); let slots = c.scratch_slots_per_lane(); assert_eq!(p.scratch_ops_per_hash(), k * ITERATIONS); for (file, text, dialect, kernels) in [ ("program.metal", metal_program(p, log2, LoadSource::Stored), Dialect::Metal, 1), ("program_bound.metal", metal_program_bound(p, log2), Dialect::Metal, 1), ("kernel.cu", cuda_kernel(p, mp), Dialect::Cuda, 1), ("kernel_bound.cu", cuda_kernel_bound(p, mp), Dialect::Cuda, 1), ("kernel.cl", opencl_kernel(p, mp), Dialect::OpenCl, 1), // the OpenCL bound file carries igneum_hash and igneum_hash_bound ("kernel_bound.cl", opencl_kernel_bound(p, mp), Dialect::OpenCl, 2), ] { let on_disk = read_pack(pack, file); assert_eq!(on_disk, text, "{pack}/{file}: the pack is the emitter's text"); // scr0 is the persistent control: an arena and a tag, no read-modify-write; the check holds with k = 0 scratch_text_check(&on_disk, dialect, k, slots, kernels).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")); checked += 1; } // the vectors of the pack are the CPU's let v: Value = serde_json::from_str(&read_pack(pack, "vectors.json")).unwrap(); for w in v["warps"].as_array().unwrap() { let base = w["base_nonce"].as_u64().unwrap() as u32; let got = e.hash_warp(base); for (lane, x) in w["expected"].as_array().unwrap().iter().enumerate() { let want = u64::from_str_radix(x.as_str().unwrap().trim_start_matches("0x"), 16).unwrap(); assert_eq!(got[lane], want, "{pack}: base {base} lane {lane}"); } } if k == 4 && slots == 64 { sample_metal = read_pack(pack, "program.metal"); sample_cl = read_pack(pack, "kernel.cl"); sample_cu = read_pack(pack, "kernel.cu"); sample_k = k; sample_slots = slots; } } assert_eq!(checked, packs.len() * 6); println!("static scratch check: {checked} kernels over {} scr packs", packs.len()); // The deliberate breaks (the watcher rule of CLAUDE.md: a check is trusted once it fails on a known-broken // case). Each must be caught; the message names what. assert!(sample_k == 4 && sample_slots == 64, "scr4k32 is in the pack set"); let mask = format!(" & {}u; uint4 v_", sample_slots - 1); let broken_mask = sample_metal.replacen(&mask, "; uint4 v_", 1); assert_ne!(broken_mask, sample_metal); let e = scratch_text_check(&broken_mask, Dialect::Metal, 4, 64, 1).unwrap_err(); assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); println!("break 1 (one mask dropped, Metal): {e}"); let wrong_mask = sample_metal.replace(" & 63u;", " & 127u;"); let e = scratch_text_check(&wrong_mask, Dialect::Metal, 4, 64, 1).unwrap_err(); assert!(e.contains("masked slot followed by the load: 0, expected 4"), "{e}"); println!("break 2 (mask 63 -> 127 on every RMW, Metal): {e}"); let wrong_stride = sample_metal.replace("* 256u;", "* 128u;"); let e = scratch_text_check(&wrong_stride, Dialect::Metal, 4, 64, 1).unwrap_err(); assert!(e.contains("arena definition: 0, expected 1"), "{e}"); println!("break 3 (arena stride 256 -> 128 words, Metal): {e}"); let stray = format!("{sample_metal}\n// stray\n// arena[0] = 0u; scratch[1] = 1u;\n"); let e = scratch_text_check(&stray, Dialect::Metal, 4, 64, 1).unwrap_err(); assert!(e.contains("arena mentions") && e.contains("direct scratch indexing: 1, expected 0"), "{e}"); println!("break 4 (a stray arena and scratch access, Metal): {e}"); let e = scratch_text_check(&sample_cl.replacen(" & 63u; uint4 v_ = vload4", "; uint4 v_ = vload4", 1), Dialect::OpenCl, 4, 64, 1).unwrap_err(); assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); println!("break 5 (one mask dropped, OpenCL): {e}"); let e = scratch_text_check(&sample_cu.replacen(" & 63u; uint4 v_ = *(const uint4*)", "; uint4 v_ = *(const uint4*)", 1), Dialect::Cuda, 4, 64, 1).unwrap_err(); assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); println!("break 6 (one mask dropped, CUDA): {e}"); // and the unbroken texts pass under the same calls scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 1).unwrap(); scratch_text_check(&sample_cl, Dialect::OpenCl, 4, 64, 1).unwrap(); scratch_text_check(&sample_cu, Dialect::Cuda, 4, 64, 1).unwrap(); // a wrong slot count, RMW count or kernel count against a right text fails too (the check is tied to the class) assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 256, 1).is_err()); assert!(scratch_text_check(&sample_metal, Dialect::Metal, 3, 64, 1).is_err()); assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 2).is_err()); } // --------------------------------------------------------------------------------------------------------------- // 5. The fuzz: 200 generated scratch programs, contract on every instruction, 4 units each across the 32-bit // range including the wrap; with IGNEUM_SCRATCH_PACKS_OUT the packs for the Metal runs (question 3, 4) // --------------------------------------------------------------------------------------------------------------- /// Write a pack whose vectors.json carries `bases` (any number of units) instead of the three standard bases. fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) -> Vec<[u64; 32]> { let mut pack = export_pack(e, day, source); let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect(); let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true); for f in pack.files.iter_mut() { if f.0 == "vectors.json" { f.1 = vj.clone(); } } pack.write_to(dir).unwrap(); outs } fn contract(p: &Program) { assert_eq!(p.instrs.len(), INSTR_COUNT); assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count() + p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), 16); assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), p.class.scratch_slots()); assert!(p.instrs[0].op != Op::Load && p.instrs[0].op != Op::Scratch, "instruction 0 is never a memory op"); for (k, i) in p.instrs.iter().enumerate() { assert!(i.src != i.dst, "#{k}: src == dst"); assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot); assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask); assert!(i.dst < 8 && i.src < 8 && i.src2 < 8); assert_eq!(i.width, 1, "#{k}: a scratch class reads one-word loads"); } assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program"); } #[test] fn fuzz_scr_programs_cpu() { let n: usize = std::env::var("IGNEUM_SCRATCH_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200); let out = std::env::var("IGNEUM_SCRATCH_PACKS_OUT").ok().map(PathBuf::from); let mut rng = SplitMix64::new(0x6967_6e65_756d_2d73); // "igneum-s" let day = "2026-10-03"; let closed = DatasetSource::new(day, DatasetMode::ClosedForm, 28); // memory-hard sources per size, built once each (the cache fill is 0.2 s); only when packs are written let mut mh: HashMap = HashMap::new(); let mut manifest = String::from("pack\tclass\tlog2\tprogram_id\tscratch_ops_per_hash\tbases\n"); let mut per_class: HashMap = HashMap::new(); let mut units = 0usize; let mut wraps = 0usize; if let Some(dir) = &out { std::fs::create_dir_all(dir).unwrap(); // the edge packs first: 64 MiB datasets (no dataset load in them), the four edge bases for kb in [32u8, 128] { for (name, _what, p) in edge_programs(kb) { let log2 = 24; let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2)); let e = Epoch { program: p, dataset: ds }; let pack_name = format!("edge-{name}-k{kb}"); write_pack_with_bases(&dir.join(&pack_name), &e, day, &EDGE_BASES, "igneum-pow tests/scratch.rs edge"); manifest.push_str(&format!( "{pack_name}\t{}\t{log2}\t{:016x}\t{}\t{}\n", e.program.class.name(), e.program.program_id(), e.program.scratch_ops_per_hash(), EDGE_BASES.iter().map(|b| format!("{b}")).collect::>().join(",") )); mh.insert(log2, e.dataset); } } } for i in 0..n { let name = CLASSES[rng.below(CLASSES.len() as u64) as usize]; let c = class(name); let seed = format!("igneum-scratch-fuzz/{i}"); let p = generate_class(&seed, c); contract(&p); *per_class.entry(name.to_string()).or_insert(0) += 1; // four bases: one inside a 256-nonce batch (in-batch check on the GPU), one straddling 2^31, one in // the last 256 nonces (the unit wraps past 2^32 or ends on it), one uniform let b0 = (rng.below(8) as u32) * 32; let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32); let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32); let b3 = (rng.next() as u32) & !31; let bases = [b0, b1, b2, b3]; // an aligned unit never straddles 2^32 (spec 1.9); the top unit ends on 0xffffffff and the persistent // kernel's unit sequence wraps inside a launch, which the Metal run checks with packbench --batch-base wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count(); // the CPU: the interpreter is deterministic and every scratch event is inside the lane's slots for &b in &bases { let (r1, ev) = interpret_warp_scratch(&p, &p.seed, b, &closed, true); let r2 = interpret_warp_scratch(&p, &p.seed, b, &closed, false).0; assert_eq!(r1.hashes, r2.hashes); assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES); assert!(ev.iter().all(|e: &ScratchEvent| e.slot < c.scratch_slots_per_lane() as u32)); units += 1; } if let Some(dir) = &out { let log2 = [24u32, 26, 28][rng.below(3) as usize]; let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2)); let e = Epoch { program: p, dataset: ds }; let pack_name = format!("fuzz-{i:03}-{name}-l{log2}"); write_pack_with_bases(&dir.join(&pack_name), &e, day, &bases, "igneum-pow tests/scratch.rs fuzz"); manifest.push_str(&format!( "{pack_name}\t{name}\t{log2}\t{:016x}\t{}\t{}\n", e.program.program_id(), e.program.scratch_ops_per_hash(), bases.iter().map(|b| format!("{b}")).collect::>().join(",") )); mh.insert(log2, e.dataset); } else { let _ = rng.below(3); } } let mut classes: Vec<_> = per_class.iter().collect(); classes.sort(); println!("fuzz: {n} programs, {units} units on the CPU, {wraps} units in the top 256 nonces, classes {classes:?}"); assert_eq!(units, 4 * n); assert_eq!(wraps, n, "every program has a unit in the top 256 nonces"); if let Some(dir) = &out { std::fs::write(dir.join("manifest.tsv"), manifest).unwrap(); println!("packs written to {}", dir.display()); } } /// The fold and rewrite, restated: a slot after `d` dependent RMWs holds 96 bits that are a function of the fill /// (3 words, a pure function of nonce, slot and seed) and the `d` fold values; a chip that keeps the `d` fold /// values (32 bits each) instead of the 96-bit slot recomputes the slot in `d` rewrites. This test pins the /// arithmetic the analysis uses (question 2): the replay from the fold values reproduces the slot. #[test] fn slot_is_replayable_from_its_fold_values() { let seed = seed_words_from_bytes(b"igneum-genesis"); let (base, lane, slot) = (0x1234_5600u32, 5u32, 17u32); let fill = [scratch_fill(&seed, base, lane, slot, 0), scratch_fill(&seed, base, lane, slot, 1), scratch_fill(&seed, base, lane, slot, 2)]; let mut rng = SplitMix64::new(99); let dsts: Vec = (0..64).map(|_| rng.next() as u32).collect(); // the honest sequence: read, fold, rewrite, 64 times let mut w = fill; let mut xs = Vec::new(); for &d in &dsts { let x = fold_words(d, &w); xs.push(x); w = scratch_rewrite(x, &w); } // the replay: from the fill and the stored fold values alone let mut w2 = fill; for &x in &xs { w2 = scratch_rewrite(x, &w2); } assert_eq!(w, w2); // and nothing shorter: the fold value at step d depends on the slot content at step d, which depends on // every earlier fold value (drop one and the chain diverges) let mut w3 = fill; for (i, &x) in xs.iter().enumerate() { if i != 10 { w3 = scratch_rewrite(x, &w3); } } assert_ne!(w, w3); }