igneum/igneum-pow/tests/scratch.rs
igneum-labs 51dba1ce76 Counter ASIC 3.0 item 8: the latency-shadow knob (LoadClass +sh<S>x<R>), its packs and the PC 2 playbook
A shadow block of S ALU instructions run R times at the end of every iteration, drawn from the program stream after
the 64 base instructions, behind LoadClass::shadow: v2 and v3 draw nothing and emit nothing (the pinned packs are
byte-identical, cargo test 54 + 4 + 19 + 7 green). The interpreter, the three kernel dialects (both kernels each),
program.h and program.json carry it; the acceptance rule interprets the base program only. Packs for seed
igneum-genesis over class mx8 at 4,096 to 180,224 shadow instructions per hash (proto-cuda/packs-ca3-shadow), and
the PC 2 bench playbook tools/ca3-shadow/pc2-shadow-bench.ps1 (passes the publisher's three checks).

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-06 07:58:35 +00:00

767 lines
42 KiB
Rust

//! Soundness tests of layer 3 of `docs/plans/counter-asic-2.md`: the per-warp scratch with read-modify-writes
//! (variant 5 of the read-width experiment, `LoadClass::scratch(k, kb)`). Analysis and results:
//! `docs/analysis/scratch-soundness.md`. Every test is parametric over the class's slot count
//! (`scratch_slots_per_lane()`), so the 32 and 128 KiB geometries and any later one run the same checks.
//!
//! What runs under plain `cargo test`:
//! 1. `rewrite_is_a_bijection_of_the_fold_value`, `fill_is_a_bijection_of_the_nonce`: the written words as
//! functions (question 1).
//! 2. `written_words_unbiased_and_rehit_rates`: bit bias of every written word over 2^11 units x 3 seeds per class
//! (the TESTS.md section 3 shape), and the measured slot re-hit rate against the birthday formula (question 2).
//! 3. `edge_programs_match_the_hand_model`: hand-built programs that drive every read-modify-write of a hash to
//! slot 0, slot MASK, through out-of-range registers, to one slot per lane, alternating two slots, and 16
//! read-modify-writes per iteration on one slot; the interpreter against an independent hand model, and the
//! hand model shown to have teeth (question 3, CPU half).
//! 4. `scr_packs_regenerate_and_pass_the_static_scratch_check`: every emitted kernel of every scr pack under
//! `proto-cuda/packs-readwidth` regenerates from its program.json and passes the static scratch-mask check;
//! the check is shown to fail on four deliberate breaks (question 4).
//! 5. `fuzz_scr_programs_cpu`: 200 generated scratch programs over the six classes, generator contract on every
//! instruction, 4 units each at base nonces across the 32-bit range including the wrap; with
//! `IGNEUM_SCRATCH_PACKS_OUT=<dir>` it also writes the packs (and the edge packs) for the Metal runs of
//! `proto-metal/packbench` (question 3 GPU half, question 4, `TESTS.md` section 9 shape).
use igneum_pow::emit::{
cuda_kernel, cuda_kernel_bound, export_pack, metal_program, metal_program_bound, opencl_kernel,
opencl_kernel_bound, vectors_json, LoadSource,
};
use igneum_pow::generator::{
generate_class, generate_from_seed_bytes_class, Instr, LoadClass, Op, Program, GENERATOR_VERSION, INSTR_COUNT,
ITERATIONS, LANES,
};
use igneum_pow::seed::{seed_words_from_bytes, SplitMix64};
use igneum_pow::verify::{
fold_words, interpret_warp_scratch, scratch_fill, scratch_rewrite, splitmix32, DatasetMode, DatasetSource,
Epoch, ScratchEvent, FOLD_MUL, FOLD_ROT,
};
use serde_json::Value;
use std::collections::HashMap;
use std::path::PathBuf;
/// The classes under study: the two capped geometries (32 and 128 KiB per warp: 64 and 256 slots per lane) at the
/// RMW shares the readwidth branch measures.
const CLASSES: [&str; 6] = ["scr2k32", "scr4k32", "scr8k32", "scr2k128", "scr4k128", "scr8k128"];
fn class(name: &str) -> LoadClass {
LoadClass::parse(name).unwrap_or_else(|| panic!("class {name}"))
}
// ---------------------------------------------------------------------------------------------------------------
// 1. The written words as functions (question 1)
// ---------------------------------------------------------------------------------------------------------------
/// For a fixed slot content `w`, each of the three rewritten words is a bijection of the fold value `x`
/// (`x ^ w1`, `rotl(x, 7) ^ w2`, `x + w0`), so the rewrite is injective in `x` and a uniform `x` gives a uniform
/// word in every position. Checked over 2^16 consecutive `x` for 16 random `w`.
#[test]
fn rewrite_is_a_bijection_of_the_fold_value() {
let mut rng = SplitMix64::new(0x7363_7261_7463_6801);
for _ in 0..16 {
let w = [rng.next() as u32, rng.next() as u32, rng.next() as u32];
let x0 = rng.next() as u32;
let mut seen = [vec![false; 1 << 16], vec![false; 1 << 16], vec![false; 1 << 16]];
for i in 0..(1u32 << 16) {
let x = x0.wrapping_add(i);
let out = scratch_rewrite(x, &w);
for j in 0..3 {
// a bijection of x maps 2^16 consecutive x to 2^16 distinct words; the low 16 bits alone are
// distinct for the xor words (x ^ c) and for the add word (x + c), since both act on the low 16
// bits as bijections of the low 16 bits of x; the rotl word is checked on its rotated-back bits
let key = if j == 1 { out[j].rotate_right(7) & 0xffff } else { out[j] & 0xffff };
assert!(!seen[j][key as usize], "word {j} repeats inside 2^16 consecutive x");
seen[j][key as usize] = true;
}
}
}
// The rewrite inverts: from the old content and any ONE written word the fold value is recovered, so a
// rewritten slot carries exactly 32 bits of new state (the point of question 2's arithmetic).
let w = [0x1234_5678, 0x9abc_def0, 0x0fed_cba9];
let x = 0xdead_beef;
let out = scratch_rewrite(x, &w);
assert_eq!(out[0] ^ w[1], x);
assert_eq!((out[1] ^ w[2]).rotate_right(7), x);
assert_eq!(out[2].wrapping_sub(w[0]), x);
}
/// For a fixed (seed, slot, j) the fill is a bijection of the lane nonce: `splitmix32` is a bijection of its
/// 32-bit input and the input `((base + lane) ^ s) + c` is a bijection of `base + lane`. Over 2^16 consecutive
/// nonces no fill word repeats, for 8 slots x 3 words.
#[test]
fn fill_is_a_bijection_of_the_nonce() {
let seed = seed_words_from_bytes(b"igneum-genesis");
for slot in [0u32, 1, 63, 64, 255, 1023, 2047] {
for j in 0..3u32 {
let mut words: Vec<u32> = (0..(1u32 << 16)).map(|n| scratch_fill(&seed, n, 0, slot, j)).collect();
words.sort_unstable();
words.dedup();
assert_eq!(words.len(), 1 << 16, "slot {slot} word {j}: fill words of 2^16 consecutive nonces are distinct");
}
}
// base + lane is the lane nonce: the fill of lane l at base b is the fill of lane 0 at base b + l
assert_eq!(scratch_fill(&seed, 0x1000, 7, 5, 2), scratch_fill(&seed, 0x1007, 0, 5, 2));
// and it wraps with the nonce: base 0xffffffe0, lane 31 is nonce 0xffffffff; lane 32 would be nonce 0
assert_eq!(scratch_fill(&seed, 0xffff_ffe0, 32, 5, 2), scratch_fill(&seed, 0, 0, 5, 2));
// the three word positions of one slot and nonce are three different permutation outputs
let f: Vec<u32> = (0..3).map(|j| scratch_fill(&seed, 12345, 7, 17, j)).collect();
assert!(f[0] != f[1] && f[1] != f[2] && f[0] != f[2]);
}
// ---------------------------------------------------------------------------------------------------------------
// 2. Uniformity of the written words and the slot re-hit rate (questions 1 and 2)
// ---------------------------------------------------------------------------------------------------------------
/// Birthday arithmetic: the expected number of distinct slots after `n` uniform draws from `s` slots.
fn expected_distinct(s: usize, n: usize) -> f64 {
let s = s as f64;
s * (1.0 - (1.0 - 1.0 / s).powi(n as i32))
}
struct ClassStats {
units: usize,
events: usize,
hits: usize,
/// ones count per bit of the written words, 3 x 32
ones: [[u64; 32]; 3],
/// ones count per bit of written XOR read (the change the rewrite makes to the slot)
delta_ones: [[u64; 32]; 3],
/// re-hit depth histogram: how many earlier RMWs the slot had seen in this unit (0 = first touch)
depth: Vec<usize>,
max_depth: usize,
/// how often each slot index was addressed (the slot comes from a register's low bits)
slot_hist: Vec<u64>,
}
fn class_stats(name: &str, seeds: &[&str], units_per_seed: usize) -> ClassStats {
let c = class(name);
let mut st = ClassStats {
units: 0,
events: 0,
hits: 0,
ones: [[0; 32]; 3],
delta_ones: [[0; 32]; 3],
depth: vec![0; 256],
max_depth: 0,
slot_hist: vec![0; c.scratch_slots_per_lane()],
};
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28);
for seed in seeds {
let p = generate_class(seed, c);
assert_eq!(p.scratch_ops_per_hash(), c.scratch_slots() * ITERATIONS);
for u in 0..units_per_seed {
let base = (u as u32).wrapping_mul(32).wrapping_add(0x4000_0000);
let (_, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true);
assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES);
let mut count: HashMap<(u8, u32), usize> = HashMap::new();
for e in &ev {
assert!(e.slot < c.scratch_slots_per_lane() as u32, "slot inside the lane's scratch");
let d = count.entry((e.lane, e.slot)).or_insert(0);
assert_eq!(e.hit, *d > 0, "hit flag agrees with the unit's own history");
assert_eq!(e.written, scratch_rewrite(e.x, &e.read));
if !e.hit {
let fill = [
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 0),
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 1),
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 2),
];
assert_eq!(e.read, fill, "a first touch reads the fill");
}
st.depth[(*d).min(255)] += 1;
st.max_depth = st.max_depth.max(*d);
st.slot_hist[e.slot as usize] += 1;
*d += 1;
st.events += 1;
st.hits += e.hit as usize;
for j in 0..3 {
for b in 0..32 {
st.ones[j][b] += ((e.written[j] >> b) & 1) as u64;
st.delta_ones[j][b] += (((e.written[j] ^ e.read[j]) >> b) & 1) as u64;
}
}
}
st.units += 1;
}
}
st
}
/// Bit bias of every written word (and of the change each rewrite makes) within 6 sigma of a fair coin, over
/// 3 seeds x 2^11 units per class (131,072 hashes per seed set); the slot re-hit rate against the birthday
/// formula within 3 percent relative. The table printed here is the one in the analysis.
#[test]
fn written_words_unbiased_and_rehit_rates() {
let seeds = ["igneum-genesis", "igneum-genesis/stats1", "igneum-genesis/stats2"];
let units = 1usize << 11;
println!("class | slots/lane | RMW/hash | events | re-hits | re-hit % | birthday % | slot chi2 z (spread) | max depth | max bias sigma | max delta bias sigma");
for name in CLASSES {
let c = class(name);
let st = class_stats(name, &seeds, units);
let n = st.events as f64;
let sigma = (n / 4.0).sqrt();
let mut worst = 0.0f64;
let mut worst_delta = 0.0f64;
for j in 0..3 {
for b in 0..32 {
let z = (st.ones[j][b] as f64 - n / 2.0).abs() / sigma;
let zd = (st.delta_ones[j][b] as f64 - n / 2.0).abs() / sigma;
assert!(z <= 6.0, "{name}: written word {j} bit {b} biased: {z:.2} sigma");
assert!(zd <= 6.0, "{name}: rewrite delta word {j} bit {b} biased: {zd:.2} sigma");
worst = worst.max(z);
worst_delta = worst_delta.max(zd);
}
}
let per_lane_hash = c.scratch_slots() * ITERATIONS;
let s = c.scratch_slots_per_lane();
let exp_hits = per_lane_hash as f64 - expected_distinct(s, per_lane_hash);
let exp_pct = 100.0 * exp_hits / per_lane_hash as f64;
let got_pct = 100.0 * st.hits as f64 / st.events as f64;
// chi-square of the slot histogram against uniform (df = s - 1): the slot is a register's low bits, and
// the measured re-hit rate runs above the uniform birthday rate (the finding of the analysis, question 2)
let expect_per_slot = n / s as f64;
let chi2: f64 = st.slot_hist.iter().map(|&h| (h as f64 - expect_per_slot).powi(2) / expect_per_slot).sum();
let chi2_z = (chi2 - (s as f64 - 1.0)) / (2.0 * (s as f64 - 1.0)).sqrt();
let hot = *st.slot_hist.iter().max().unwrap() as f64 / expect_per_slot;
let cold = *st.slot_hist.iter().min().unwrap() as f64 / expect_per_slot;
println!(
"{name} | {s} | {per_lane_hash} | {} | {} | {got_pct:.2} | {exp_pct:.2} | {chi2_z:.1} (hottest slot {hot:.2}x, coldest {cold:.2}x) | {} | {worst:.2} | {worst_delta:.2}",
st.events, st.hits, st.max_depth
);
// a regression band, not a uniformity claim: the rate sits between the uniform birthday rate and twice it
assert!(
got_pct >= 0.9 * exp_pct && got_pct <= 2.0 * exp_pct,
"{name}: re-hit rate {got_pct:.2}% against birthday {exp_pct:.2}%"
);
// depth histogram: the number of earlier RMWs a re-hit slot had seen in the unit
let shown: Vec<String> = st.depth.iter().take(st.max_depth + 1).enumerate().map(|(d, n)| format!("{d}:{n}")).collect();
println!(" depth histogram {}", shown.join(" "));
}
}
// ---------------------------------------------------------------------------------------------------------------
// 3. Hand-built edge programs against an independent hand model (question 3, CPU half)
// ---------------------------------------------------------------------------------------------------------------
fn ins(op: Op, dst: u8, src: u8) -> Instr {
Instr { op, dst, src, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 }
}
fn add_imm(dst: u8, src: u8, imm: u32) -> Instr {
Instr { op: Op::Add, dst, src, src2: 0, imm, imm2: imm, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 }
}
/// A hand-built program of class `c` named `name` (its seed is the name, so its fill words and init words are
/// its own). These bypass the generator and the acceptance rule, like `TESTS.md` section 2; `sub r, r` zeroes a
/// register as the Swift edge set does.
fn edge(name: &str, c: LoadClass, instrs: Vec<Instr>) -> Program {
let seed_string = format!("igneum-scratch-edge/{name}");
let seed_bytes = seed_string.as_bytes().to_vec();
let k = instrs.iter().filter(|i| i.op == Op::Scratch).count();
assert_eq!(k, c.scratch_slots(), "{name}: the class carries the program's scratch count");
Program {
seed: seed_words_from_bytes(&seed_bytes),
seed_string,
seed_bytes,
generator: GENERATOR_VERSION,
attempt: 0,
class: c,
era_bytes: None,
instrs,
shadow: Vec::new(),
}
}
/// The edge set for a scratch of `kb` KiB per warp. Each entry: (name, what it drives, program).
fn edge_programs(kb: u8) -> Vec<(String, &'static str, Program)> {
let m = LoadClass::scratch(1, kb).scratch_slot_mask();
let dsts = [2u8, 3, 4, 5, 6, 7, 0, 2, 3, 4, 5, 6, 7, 0, 2, 3];
let scr = |n: usize, src: u8| -> Vec<Instr> { (0..n).map(|i| ins(Op::Scratch, dsts[i], src)).collect() };
let mut v = Vec::new();
// every RMW of the hash to slot 0 through a zero register: 64 dependent RMWs on one slot per lane
let mut p = vec![ins(Op::Sub, 1, 1)];
p.extend(scr(8, 1));
v.push(("slot0".to_string(), "r1 = 0: every RMW to slot 0", edge(&format!("slot0/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot MASK through the in-range register MASK
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m)];
p.extend(scr(8, 1));
v.push(("slotmask".to_string(), "r1 = MASK: every RMW to the last slot", edge(&format!("slotmask/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot MASK through the out-of-range register 0xffffffff
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, 1), ins(Op::Sub, 1, 2)];
p.extend(scr(8, 1));
v.push(("ones".to_string(), "r1 = 0xffffffff: masked to the last slot", edge(&format!("ones/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot 0 through the out-of-range register MASK + 1
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m.wrapping_add(1))];
p.extend(scr(8, 1));
v.push(("maskplus1".to_string(), "r1 = MASK + 1: masked to slot 0", edge(&format!("maskplus1/k{kb}"), LoadClass::scratch(8, kb), p)));
// 16 RMWs per iteration on slot 0: 128 dependent RMWs on one slot per lane per hash
let mut p = vec![ins(Op::Sub, 1, 1)];
p.extend(scr(16, 1));
v.push(("sixteen".to_string(), "16 RMWs per iteration on slot 0", edge(&format!("sixteen/k{kb}"), LoadClass::scratch(16, kb), p)));
// one slot per lane from the init words: lanes with equal slots would show any cross-lane aliasing
// (r5 is the slot register and is never a destination here)
let p: Vec<Instr> = [0u8, 1, 2, 3, 4, 6, 7, 0].iter().map(|&d| ins(Op::Scratch, d, 5)).collect();
v.push(("lanevar".to_string(), "r5 never written: one init-dependent slot per lane", edge(&format!("lanevar/k{kb}"), LoadClass::scratch(8, kb), p)));
// alternating slot 0 and slot MASK inside one iteration
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, m)];
for (i, &d) in [3u8, 4, 5, 6, 7, 0, 3, 4].iter().enumerate() {
// r1 and r2 hold the two slots and are never destinations
p.push(ins(Op::Scratch, d, if i % 2 == 0 { 1 } else { 2 }));
}
v.push(("twoslots".to_string(), "slot 0 and slot MASK alternating", edge(&format!("twoslots/k{kb}"), LoadClass::scratch(8, kb), p)));
v
}
/// The hand model: a second, minimal interpreter for the ops the edge programs use (sub, add, scratch), with its
/// own slot store keyed by (lane, slot). `mutate` swaps the rewrite's words to show the comparison has teeth.
fn hand_model(p: &Program, base: u32, mutate: bool) -> [u64; 32] {
let seed = &p.seed;
let m = p.class.scratch_slot_mask();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
let nonce = base.wrapping_add(lane as u32);
for i in 0..8 {
let mut x = nonce ^ seed[i];
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
x = splitmix32(x);
r[i][lane] = x ^ seed[(i + 1) & 7];
}
}
let mut store: HashMap<(usize, u32), [u32; 3]> = HashMap::new();
for _ in 0..ITERATIONS {
let sel = r[0];
for ins in &p.instrs {
let (d, a) = (ins.dst as usize, ins.src as usize);
match ins.op {
Op::Sub => {
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_sub(r[a][lane]);
}
}
Op::Add => {
for lane in 0..LANES {
let c = if (sel[lane] >> ins.bit) & 1 != 0 { ins.imm2 } else { ins.imm };
r[d][lane] = r[d][lane].wrapping_add(r[a][lane]).wrapping_add(c);
}
}
Op::Scratch => {
for lane in 0..LANES {
let slot = r[a][lane] & m;
let w = *store.entry((lane, slot)).or_insert_with(|| {
let mut f = [0u32; 3];
for j in 0..3u32 {
// the fill, written out in full rather than through verify::scratch_fill
let n = base.wrapping_add(lane as u32);
f[j as usize] = splitmix32(
(n ^ seed[j as usize])
.wrapping_add(slot.wrapping_mul(0x9E37_79B1))
.wrapping_add((j + 1).wrapping_mul(0x85EB_CA77)),
);
}
f
});
let mut x = r[d][lane] ^ w[0];
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[1];
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[2];
r[d][lane] = x;
let out = if mutate {
[x.rotate_left(7) ^ w[2], x ^ w[1], x.wrapping_add(w[0])]
} else {
[x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])]
};
store.insert((lane, slot), out);
}
}
other => panic!("the hand model does not implement {other:?}"),
}
}
}
let mut out = [0u64; 32];
for lane in 0..LANES {
let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21);
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
out[lane] = ((hi as u64) << 32) | lo as u64;
}
out
}
/// The four unit bases of every edge vector: 0 and 32 (two consecutive units, the pair a one-warp persistent
/// launch runs on one arena), a unit straddling 2^31, and the unit that wraps past 2^32.
const EDGE_BASES: [u32; 4] = [0, 32, 0x7fff_fff0, 0xffff_ffe0];
#[test]
fn edge_programs_match_the_hand_model() {
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24);
let mut cases = 0;
for kb in [32u8, 128] {
for (name, what, p) in edge_programs(kb) {
let slots = p.class.scratch_slots_per_lane();
for base in EDGE_BASES {
let (res, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true);
let hand = hand_model(&p, base, false);
assert_eq!(res.hashes, hand, "{name} k{kb} base {base:#x}: interpreter against the hand model ({what})");
assert_ne!(res.hashes, hand_model(&p, base, true), "{name} k{kb}: the comparison has teeth");
// the slots the trace saw are the ones the program was built to drive
let slot_set: std::collections::BTreeSet<u32> = ev.iter().map(|e| e.slot).collect();
let m = (slots - 1) as u32;
match name.as_str() {
"slot0" | "maskplus1" | "sixteen" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![0]),
"slotmask" | "ones" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![m]),
"twoslots" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![0, m]),
"lanevar" => {
for e in &ev {
assert!(e.slot <= m);
}
}
_ => unreachable!(),
}
// the chain depth on the driven slot: every RMW after the first per lane is a re-hit
let per_lane = p.scratch_ops_per_hash();
let hits = ev.iter().filter(|e| e.hit).count();
let expected_hits = match name.as_str() {
"twoslots" => (per_lane - 2) * LANES,
_ => (per_lane - 1) * LANES,
};
assert_eq!(hits, expected_hits, "{name} k{kb}: re-hits");
cases += 1;
}
}
}
assert_eq!(cases, 2 * 7 * 4);
}
// ---------------------------------------------------------------------------------------------------------------
// 4. The static scratch check over every emitted kernel of every scr pack (question 4)
// ---------------------------------------------------------------------------------------------------------------
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum Dialect {
Metal,
Cuda,
OpenCl,
}
/// The static scratch check: every scratch read-modify-write in an emitted kernel has the one masked form the
/// emitter writes, the arena is the lane's own `slots x 4` words, the tag is `salt + unit`, and nothing else
/// touches the scratch. Like the dataset mask check of `TESTS.md` section 5 and `tests/packs.rs`, a text check:
/// the guarantee is that the emitter has one template and it masks.
pub fn scratch_text_check(text: &str, dialect: Dialect, k: usize, slots: usize, kernels: usize) -> Result<(), String> {
assert!(kernels >= 1);
// every count below is per hash kernel; an OpenCL bound file carries igneum_hash and igneum_hash_bound
let k = k * kernels;
assert!(slots.is_power_of_two() && slots >= 1);
let mask = (slots - 1) as u32;
let wpl = slots * 4;
let (u, load, store, ptr) = match dialect {
Dialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "device uint* arena"),
Dialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "uint32_t* arena"),
Dialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); }", "__global uint* arena"),
};
let count = |needle: &str| text.matches(needle).count();
let mut errs = Vec::new();
let mut expect = |what: &str, got: usize, want: usize| {
if got != want {
errs.push(format!("{what}: {got}, expected {want}"));
}
};
// k slot computations, each masked with exactly the class's mask and immediately followed by the one load form
expect("slot definitions `{ u s_ = r`", count(&format!("{{ {u} s_ = r")), k);
expect("masked slot followed by the load", count(&format!(" & {mask}u; {load}")), k);
expect("stores of the tagged slot", count(store), k);
expect("tag compares", count("(v_.x == tag)"), k);
expect("fill calls (three per RMW)", count("scr_fill(gbase, lane, s_, "), 3 * k);
// the arena: one definition with the class's words per lane, and 2k uses (one load, one store per RMW)
expect("arena definition", count(&format!("{ptr} = scratch + ((size_t)warp_ * 32u + lane) * {wpl}u;")), kernels);
expect("arena mentions (definition + load + store per RMW)", count("arena"), kernels + 2 * k);
expect("tag definition `tag = salt + g_`", count(&format!("{u} tag = salt + g_;")), kernels);
expect("direct scratch indexing", count("scratch["), 0);
expect("scratch pointer arithmetic outside the arena definition", count("scratch +"), kernels);
// no other mask value on a slot: every `s_ = r` line carries the class mask and nothing else carries ` & Nu; uint4 v_`
let any_mask_load = count(&format!("u; {load}"));
expect("loads preceded by some mask (must all be the class mask)", any_mask_load, k);
if errs.is_empty() {
Ok(())
} else {
Err(errs.join("; "))
}
}
fn packs_rw_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-readwidth")
}
fn scr_packs() -> Vec<String> {
let mut v: Vec<String> = std::fs::read_dir(packs_rw_dir())
.unwrap()
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
.filter(|n| n.starts_with("scr"))
.collect();
v.sort();
v
}
fn read_pack(pack: &str, file: &str) -> String {
let p = packs_rw_dir().join(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
/// Every scr pack regenerates from its program.json (seed bytes, class, day bytes, size) to the same six kernel
/// texts, byte for byte, and every one of those texts passes the static scratch check for the class's k and slot
/// count; the check fails on four deliberate breaks of a copy of the Metal text (mask dropped, mask changed, arena
/// stride changed, a stray scratch access) and on the OpenCL and CUDA twins of the first.
#[test]
fn scr_packs_regenerate_and_pass_the_static_scratch_check() {
let packs = scr_packs();
assert!(packs.len() >= 6, "the scr packs: {packs:?}");
let mut checked = 0;
let mut sample_metal = String::new();
let mut sample_cl = String::new();
let mut sample_cu = String::new();
let mut sample_k = 0;
let mut sample_slots = 0;
for pack in &packs {
let j: Value = serde_json::from_str(&read_pack(pack, "program.json")).unwrap();
let name = j["load_class"].as_str().unwrap();
let c = class(name);
assert_eq!(&format!("{name}"), pack, "pack directory named after its class");
let seed = j["seed"].as_str().unwrap();
let seed_bytes = igneum_pow::bind::unhex(j["seed_bytes"].as_str().unwrap()).unwrap();
let day_bytes = igneum_pow::bind::unhex(j["dataset"]["day_bytes"].as_str().unwrap()).unwrap();
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard");
let program = generate_from_seed_bytes_class(seed, &seed_bytes, c);
assert_eq!(program.class, c);
assert_eq!(program.program_id(), u64::from_str_radix(j["program_id"].as_str().unwrap().trim_start_matches("0x"), 16).unwrap());
let mut dataset = DatasetSource::from_key(seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2);
dataset.key_bytes = day_bytes;
let e = Epoch { program, dataset };
let p = &e.program;
let mp = e.dataset.memhard().map(|m| &m.params);
let k = c.scratch_slots();
let slots = c.scratch_slots_per_lane();
assert_eq!(p.scratch_ops_per_hash(), k * ITERATIONS);
for (file, text, dialect, kernels) in [
("program.metal", metal_program(p, log2, LoadSource::Stored), Dialect::Metal, 1),
("program_bound.metal", metal_program_bound(p, log2), Dialect::Metal, 1),
("kernel.cu", cuda_kernel(p, mp), Dialect::Cuda, 1),
("kernel_bound.cu", cuda_kernel_bound(p, mp), Dialect::Cuda, 1),
("kernel.cl", opencl_kernel(p, mp), Dialect::OpenCl, 1),
// the OpenCL bound file carries igneum_hash and igneum_hash_bound
("kernel_bound.cl", opencl_kernel_bound(p, mp), Dialect::OpenCl, 2),
] {
let on_disk = read_pack(pack, file);
assert_eq!(on_disk, text, "{pack}/{file}: the pack is the emitter's text");
// scr0 is the persistent control: an arena and a tag, no read-modify-write; the check holds with k = 0
scratch_text_check(&on_disk, dialect, k, slots, kernels).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"));
checked += 1;
}
// the vectors of the pack are the CPU's
let v: Value = serde_json::from_str(&read_pack(pack, "vectors.json")).unwrap();
for w in v["warps"].as_array().unwrap() {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let got = e.hash_warp(base);
for (lane, x) in w["expected"].as_array().unwrap().iter().enumerate() {
let want = u64::from_str_radix(x.as_str().unwrap().trim_start_matches("0x"), 16).unwrap();
assert_eq!(got[lane], want, "{pack}: base {base} lane {lane}");
}
}
if k == 4 && slots == 64 {
sample_metal = read_pack(pack, "program.metal");
sample_cl = read_pack(pack, "kernel.cl");
sample_cu = read_pack(pack, "kernel.cu");
sample_k = k;
sample_slots = slots;
}
}
assert_eq!(checked, packs.len() * 6);
println!("static scratch check: {checked} kernels over {} scr packs", packs.len());
// The deliberate breaks (the watcher rule of CLAUDE.md: a check is trusted once it fails on a known-broken
// case). Each must be caught; the message names what.
assert!(sample_k == 4 && sample_slots == 64, "scr4k32 is in the pack set");
let mask = format!(" & {}u; uint4 v_", sample_slots - 1);
let broken_mask = sample_metal.replacen(&mask, "; uint4 v_", 1);
assert_ne!(broken_mask, sample_metal);
let e = scratch_text_check(&broken_mask, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 1 (one mask dropped, Metal): {e}");
let wrong_mask = sample_metal.replace(" & 63u;", " & 127u;");
let e = scratch_text_check(&wrong_mask, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 0, expected 4"), "{e}");
println!("break 2 (mask 63 -> 127 on every RMW, Metal): {e}");
let wrong_stride = sample_metal.replace("* 256u;", "* 128u;");
let e = scratch_text_check(&wrong_stride, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("arena definition: 0, expected 1"), "{e}");
println!("break 3 (arena stride 256 -> 128 words, Metal): {e}");
let stray = format!("{sample_metal}\n// stray\n// arena[0] = 0u; scratch[1] = 1u;\n");
let e = scratch_text_check(&stray, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("arena mentions") && e.contains("direct scratch indexing: 1, expected 0"), "{e}");
println!("break 4 (a stray arena and scratch access, Metal): {e}");
let e = scratch_text_check(&sample_cl.replacen(" & 63u; uint4 v_ = vload4", "; uint4 v_ = vload4", 1), Dialect::OpenCl, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 5 (one mask dropped, OpenCL): {e}");
let e = scratch_text_check(&sample_cu.replacen(" & 63u; uint4 v_ = *(const uint4*)", "; uint4 v_ = *(const uint4*)", 1), Dialect::Cuda, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 6 (one mask dropped, CUDA): {e}");
// and the unbroken texts pass under the same calls
scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 1).unwrap();
scratch_text_check(&sample_cl, Dialect::OpenCl, 4, 64, 1).unwrap();
scratch_text_check(&sample_cu, Dialect::Cuda, 4, 64, 1).unwrap();
// a wrong slot count, RMW count or kernel count against a right text fails too (the check is tied to the class)
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 256, 1).is_err());
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 3, 64, 1).is_err());
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 2).is_err());
}
// ---------------------------------------------------------------------------------------------------------------
// 5. The fuzz: 200 generated scratch programs, contract on every instruction, 4 units each across the 32-bit
// range including the wrap; with IGNEUM_SCRATCH_PACKS_OUT the packs for the Metal runs (question 3, 4)
// ---------------------------------------------------------------------------------------------------------------
/// Write a pack whose vectors.json carries `bases` (any number of units) instead of the three standard bases.
fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) -> Vec<[u64; 32]> {
let mut pack = export_pack(e, day, source);
let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect();
let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true);
for f in pack.files.iter_mut() {
if f.0 == "vectors.json" {
f.1 = vj.clone();
}
}
pack.write_to(dir).unwrap();
outs
}
fn contract(p: &Program) {
assert_eq!(p.instrs.len(), INSTR_COUNT);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count() + p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), 16);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), p.class.scratch_slots());
assert!(p.instrs[0].op != Op::Load && p.instrs[0].op != Op::Scratch, "instruction 0 is never a memory op");
for (k, i) in p.instrs.iter().enumerate() {
assert!(i.src != i.dst, "#{k}: src == dst");
assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot);
assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask);
assert!(i.dst < 8 && i.src < 8 && i.src2 < 8);
assert_eq!(i.width, 1, "#{k}: a scratch class reads one-word loads");
}
assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program");
}
#[test]
fn fuzz_scr_programs_cpu() {
let n: usize = std::env::var("IGNEUM_SCRATCH_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200);
let out = std::env::var("IGNEUM_SCRATCH_PACKS_OUT").ok().map(PathBuf::from);
let mut rng = SplitMix64::new(0x6967_6e65_756d_2d73); // "igneum-s"
let day = "2026-10-03";
let closed = DatasetSource::new(day, DatasetMode::ClosedForm, 28);
// memory-hard sources per size, built once each (the cache fill is 0.2 s); only when packs are written
let mut mh: HashMap<u32, DatasetSource> = HashMap::new();
let mut manifest = String::from("pack\tclass\tlog2\tprogram_id\tscratch_ops_per_hash\tbases\n");
let mut per_class: HashMap<String, usize> = HashMap::new();
let mut units = 0usize;
let mut wraps = 0usize;
if let Some(dir) = &out {
std::fs::create_dir_all(dir).unwrap();
// the edge packs first: 64 MiB datasets (no dataset load in them), the four edge bases
for kb in [32u8, 128] {
for (name, _what, p) in edge_programs(kb) {
let log2 = 24;
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2));
let e = Epoch { program: p, dataset: ds };
let pack_name = format!("edge-{name}-k{kb}");
write_pack_with_bases(&dir.join(&pack_name), &e, day, &EDGE_BASES, "igneum-pow tests/scratch.rs edge");
manifest.push_str(&format!(
"{pack_name}\t{}\t{log2}\t{:016x}\t{}\t{}\n",
e.program.class.name(),
e.program.program_id(),
e.program.scratch_ops_per_hash(),
EDGE_BASES.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
));
mh.insert(log2, e.dataset);
}
}
}
for i in 0..n {
let name = CLASSES[rng.below(CLASSES.len() as u64) as usize];
let c = class(name);
let seed = format!("igneum-scratch-fuzz/{i}");
let p = generate_class(&seed, c);
contract(&p);
*per_class.entry(name.to_string()).or_insert(0) += 1;
// four bases: one inside a 256-nonce batch (in-batch check on the GPU), one straddling 2^31, one in
// the last 256 nonces (the unit wraps past 2^32 or ends on it), one uniform
let b0 = (rng.below(8) as u32) * 32;
let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32);
let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32);
let b3 = (rng.next() as u32) & !31;
let bases = [b0, b1, b2, b3];
// an aligned unit never straddles 2^32 (spec 1.9); the top unit ends on 0xffffffff and the persistent
// kernel's unit sequence wraps inside a launch, which the Metal run checks with packbench --batch-base
wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count();
// the CPU: the interpreter is deterministic and every scratch event is inside the lane's slots
for &b in &bases {
let (r1, ev) = interpret_warp_scratch(&p, &p.seed, b, &closed, true);
let r2 = interpret_warp_scratch(&p, &p.seed, b, &closed, false).0;
assert_eq!(r1.hashes, r2.hashes);
assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES);
assert!(ev.iter().all(|e: &ScratchEvent| e.slot < c.scratch_slots_per_lane() as u32));
units += 1;
}
if let Some(dir) = &out {
let log2 = [24u32, 26, 28][rng.below(3) as usize];
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2));
let e = Epoch { program: p, dataset: ds };
let pack_name = format!("fuzz-{i:03}-{name}-l{log2}");
write_pack_with_bases(&dir.join(&pack_name), &e, day, &bases, "igneum-pow tests/scratch.rs fuzz");
manifest.push_str(&format!(
"{pack_name}\t{name}\t{log2}\t{:016x}\t{}\t{}\n",
e.program.program_id(),
e.program.scratch_ops_per_hash(),
bases.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
));
mh.insert(log2, e.dataset);
} else {
let _ = rng.below(3);
}
}
let mut classes: Vec<_> = per_class.iter().collect();
classes.sort();
println!("fuzz: {n} programs, {units} units on the CPU, {wraps} units in the top 256 nonces, classes {classes:?}");
assert_eq!(units, 4 * n);
assert_eq!(wraps, n, "every program has a unit in the top 256 nonces");
if let Some(dir) = &out {
std::fs::write(dir.join("manifest.tsv"), manifest).unwrap();
println!("packs written to {}", dir.display());
}
}
/// The fold and rewrite, restated: a slot after `d` dependent RMWs holds 96 bits that are a function of the fill
/// (3 words, a pure function of nonce, slot and seed) and the `d` fold values; a chip that keeps the `d` fold
/// values (32 bits each) instead of the 96-bit slot recomputes the slot in `d` rewrites. This test pins the
/// arithmetic the analysis uses (question 2): the replay from the fold values reproduces the slot.
#[test]
fn slot_is_replayable_from_its_fold_values() {
let seed = seed_words_from_bytes(b"igneum-genesis");
let (base, lane, slot) = (0x1234_5600u32, 5u32, 17u32);
let fill = [scratch_fill(&seed, base, lane, slot, 0), scratch_fill(&seed, base, lane, slot, 1), scratch_fill(&seed, base, lane, slot, 2)];
let mut rng = SplitMix64::new(99);
let dsts: Vec<u32> = (0..64).map(|_| rng.next() as u32).collect();
// the honest sequence: read, fold, rewrite, 64 times
let mut w = fill;
let mut xs = Vec::new();
for &d in &dsts {
let x = fold_words(d, &w);
xs.push(x);
w = scratch_rewrite(x, &w);
}
// the replay: from the fill and the stored fold values alone
let mut w2 = fill;
for &x in &xs {
w2 = scratch_rewrite(x, &w2);
}
assert_eq!(w, w2);
// and nothing shorter: the fold value at step d depends on the slot content at step d, which depends on
// every earlier fold value (drop one and the chain diverges)
let mut w3 = fill;
for (i, &x) in xs.iter().enumerate() {
if i != 10 {
w3 = scratch_rewrite(x, &w3);
}
}
assert_ne!(w, w3);
}