From 60ccd4cd07c450bc6a1d8a9002a5d310d043b067 Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Mon, 5 Oct 2026 20:13:04 +0000 Subject: [PATCH] scratch soundness (layer 3 of Counter ASIC 2.0): verify.rs scratch trace hook; tests/scratch.rs: rewrite and fill bijections, written-word bias and re-hit rates per class, hand-built slot edge programs against a hand model, static scratch-mask check over every emitted kernel of every scr pack with six deliberate breaks, 200-program CPU fuzz and pack writer for the Metal runs; packbench --batch-base for the launch-level nonce wrap Co-Authored-By: Claude Fable 5.1 --- igneum-pow/src/verify.rs | 44 ++- igneum-pow/tests/scratch.rs | 765 ++++++++++++++++++++++++++++++++++++ proto-metal/packbench.swift | 14 +- 3 files changed, 815 insertions(+), 8 deletions(-) create mode 100644 igneum-pow/tests/scratch.rs diff --git a/igneum-pow/src/verify.rs b/igneum-pow/src/verify.rs index fedfe1b9..2f32928e 100644 --- a/igneum-pow/src/verify.rs +++ b/igneum-pow/src/verify.rs @@ -51,6 +51,22 @@ pub struct ScratchModel { data: Vec<[u32; 3]>, pub reads: usize, pub writes: usize, + /// Soundness tests (`tests/scratch.rs`, `docs/analysis/scratch-soundness.md`): when `Some`, every + /// read-modify-write is appended as it happened. `None` on every verification path. + pub trace: Option>, +} + +/// One scratch read-modify-write as the interpreter saw it (variant 5 soundness tests). +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ScratchEvent { + pub lane: u8, + pub slot: u32, + /// The slot had been written earlier in this unit (a re-hit): the words read were a rewrite, not the fill. + pub hit: bool, + pub read: [u32; 3], + /// The fold result, the new value of `dst`. + pub x: u32, + pub written: [u32; 3], } impl ScratchModel { @@ -61,6 +77,7 @@ impl ScratchModel { data: vec![[0; 3]; LANES * slots_per_lane], reads: 0, writes: 0, + trace: None, } } /// Read slot `slot` of `lane`, then rewrite it from the fold result `x`. Returns the three words read. @@ -77,7 +94,11 @@ impl ScratchModel { ] }; let x = fold_words(dst, &w); - self.data[i] = scratch_rewrite(x, &w); + let out = scratch_rewrite(x, &w); + if let Some(t) = self.trace.as_mut() { + t.push(ScratchEvent { lane: lane as u8, slot, hit: self.written[i], read: w, x, written: out }); + } + self.data[i] = out; self.written[i] = true; self.reads += 1; self.writes += 1; @@ -243,6 +264,19 @@ pub fn interpret_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) -> /// [`interpret_warp`] with explicit init words `I` (section 1.6 of the spec). The packs use `I = program.seed`; /// a block uses `I = bind::block_init_words(H, nonce)`. pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, ds: &DatasetSource) -> WarpResult { + interpret_warp_scratch(program, seed, base_nonce, ds, false).0 +} + +/// [`interpret_warp_init`] that also returns every scratch read-modify-write of the unit in execution order +/// (lane-minor within an instruction, as the interpreter runs them) when `trace` is set; empty otherwise and for +/// a class without a scratch. For the soundness tests of variant 5 only. +pub fn interpret_warp_scratch( + program: &Program, + seed: &[u32; 8], + base_nonce: u32, + ds: &DatasetSource, + trace: bool, +) -> (WarpResult, Vec) { let mask = ds.mask; let mut r = [[0u32; LANES]; 8]; for lane in 0..LANES { @@ -258,6 +292,11 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, let mut idx = [0u32; LANES]; let mut val = [0u32; LANES]; let mut scratch = if program.has_scratch() { Some(ScratchModel::new(program.class.scratch_slots_per_lane())) } else { None }; + if trace { + if let Some(m) = scratch.as_mut() { + m.trace = Some(Vec::new()); + } + } let slot_mask = program.class.scratch_slot_mask(); for _ in 0..ITERATIONS { let sel = r[0]; @@ -279,7 +318,8 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27); hashes[lane] = ((hi as u64) << 32) | lo as u64; } - WarpResult { hashes, items_derived } + let events = scratch.and_then(|m| m.trace).unwrap_or_default(); + (WarpResult { hashes, items_derived }, events) } #[inline(always)] diff --git a/igneum-pow/tests/scratch.rs b/igneum-pow/tests/scratch.rs new file mode 100644 index 00000000..d00bbfe4 --- /dev/null +++ b/igneum-pow/tests/scratch.rs @@ -0,0 +1,765 @@ +//! Soundness tests of layer 3 of `docs/plans/counter-asic-2.md`: the per-warp scratch with read-modify-writes +//! (variant 5 of the read-width experiment, `LoadClass::scratch(k, kb)`). Analysis and results: +//! `docs/analysis/scratch-soundness.md`. Every test is parametric over the class's slot count +//! (`scratch_slots_per_lane()`), so the 32 and 128 KiB geometries and any later one run the same checks. +//! +//! What runs under plain `cargo test`: +//! 1. `rewrite_is_a_bijection_of_the_fold_value`, `fill_is_a_bijection_of_the_nonce`: the written words as +//! functions (question 1). +//! 2. `written_words_unbiased_and_rehit_rates`: bit bias of every written word over 2^11 units x 3 seeds per class +//! (the TESTS.md section 3 shape), and the measured slot re-hit rate against the birthday formula (question 2). +//! 3. `edge_programs_match_the_hand_model`: hand-built programs that drive every read-modify-write of a hash to +//! slot 0, slot MASK, through out-of-range registers, to one slot per lane, alternating two slots, and 16 +//! read-modify-writes per iteration on one slot; the interpreter against an independent hand model, and the +//! hand model shown to have teeth (question 3, CPU half). +//! 4. `scr_packs_regenerate_and_pass_the_static_scratch_check`: every emitted kernel of every scr pack under +//! `proto-cuda/packs-readwidth` regenerates from its program.json and passes the static scratch-mask check; +//! the check is shown to fail on four deliberate breaks (question 4). +//! 5. `fuzz_scr_programs_cpu`: 200 generated scratch programs over the six classes, generator contract on every +//! instruction, 4 units each at base nonces across the 32-bit range including the wrap; with +//! `IGNEUM_SCRATCH_PACKS_OUT=` it also writes the packs (and the edge packs) for the Metal runs of +//! `proto-metal/packbench` (question 3 GPU half, question 4, `TESTS.md` section 9 shape). + +use igneum_pow::emit::{ + cuda_kernel, cuda_kernel_bound, export_pack, metal_program, metal_program_bound, opencl_kernel, + opencl_kernel_bound, vectors_json, LoadSource, +}; +use igneum_pow::generator::{ + generate_class, generate_from_seed_bytes_class, Instr, LoadClass, Op, Program, GENERATOR_VERSION, INSTR_COUNT, + ITERATIONS, LANES, +}; +use igneum_pow::seed::{seed_words_from_bytes, SplitMix64}; +use igneum_pow::verify::{ + fold_words, interpret_warp_scratch, scratch_fill, scratch_rewrite, splitmix32, DatasetMode, DatasetSource, + Epoch, ScratchEvent, FOLD_MUL, FOLD_ROT, +}; +use serde_json::Value; +use std::collections::HashMap; +use std::path::PathBuf; + +/// The classes under study: the two capped geometries (32 and 128 KiB per warp: 64 and 256 slots per lane) at the +/// RMW shares the readwidth branch measures. +const CLASSES: [&str; 6] = ["scr2k32", "scr4k32", "scr8k32", "scr2k128", "scr4k128", "scr8k128"]; + +fn class(name: &str) -> LoadClass { + LoadClass::parse(name).unwrap_or_else(|| panic!("class {name}")) +} + +// --------------------------------------------------------------------------------------------------------------- +// 1. The written words as functions (question 1) +// --------------------------------------------------------------------------------------------------------------- + +/// For a fixed slot content `w`, each of the three rewritten words is a bijection of the fold value `x` +/// (`x ^ w1`, `rotl(x, 7) ^ w2`, `x + w0`), so the rewrite is injective in `x` and a uniform `x` gives a uniform +/// word in every position. Checked over 2^16 consecutive `x` for 16 random `w`. +#[test] +fn rewrite_is_a_bijection_of_the_fold_value() { + let mut rng = SplitMix64::new(0x7363_7261_7463_6801); + for _ in 0..16 { + let w = [rng.next() as u32, rng.next() as u32, rng.next() as u32]; + let x0 = rng.next() as u32; + let mut seen = [vec![false; 1 << 16], vec![false; 1 << 16], vec![false; 1 << 16]]; + for i in 0..(1u32 << 16) { + let x = x0.wrapping_add(i); + let out = scratch_rewrite(x, &w); + for j in 0..3 { + // a bijection of x maps 2^16 consecutive x to 2^16 distinct words; the low 16 bits alone are + // distinct for the xor words (x ^ c) and for the add word (x + c), since both act on the low 16 + // bits as bijections of the low 16 bits of x; the rotl word is checked on its rotated-back bits + let key = if j == 1 { out[j].rotate_right(7) & 0xffff } else { out[j] & 0xffff }; + assert!(!seen[j][key as usize], "word {j} repeats inside 2^16 consecutive x"); + seen[j][key as usize] = true; + } + } + } + // The rewrite inverts: from the old content and any ONE written word the fold value is recovered, so a + // rewritten slot carries exactly 32 bits of new state (the point of question 2's arithmetic). + let w = [0x1234_5678, 0x9abc_def0, 0x0fed_cba9]; + let x = 0xdead_beef; + let out = scratch_rewrite(x, &w); + assert_eq!(out[0] ^ w[1], x); + assert_eq!((out[1] ^ w[2]).rotate_right(7), x); + assert_eq!(out[2].wrapping_sub(w[0]), x); +} + +/// For a fixed (seed, slot, j) the fill is a bijection of the lane nonce: `splitmix32` is a bijection of its +/// 32-bit input and the input `((base + lane) ^ s) + c` is a bijection of `base + lane`. Over 2^16 consecutive +/// nonces no fill word repeats, for 8 slots x 3 words. +#[test] +fn fill_is_a_bijection_of_the_nonce() { + let seed = seed_words_from_bytes(b"igneum-genesis"); + for slot in [0u32, 1, 63, 64, 255, 1023, 2047] { + for j in 0..3u32 { + let mut words: Vec = (0..(1u32 << 16)).map(|n| scratch_fill(&seed, n, 0, slot, j)).collect(); + words.sort_unstable(); + words.dedup(); + assert_eq!(words.len(), 1 << 16, "slot {slot} word {j}: fill words of 2^16 consecutive nonces are distinct"); + } + } + // base + lane is the lane nonce: the fill of lane l at base b is the fill of lane 0 at base b + l + assert_eq!(scratch_fill(&seed, 0x1000, 7, 5, 2), scratch_fill(&seed, 0x1007, 0, 5, 2)); + // and it wraps with the nonce: base 0xffffffe0, lane 31 is nonce 0xffffffff; lane 32 would be nonce 0 + assert_eq!(scratch_fill(&seed, 0xffff_ffe0, 32, 5, 2), scratch_fill(&seed, 0, 0, 5, 2)); + // the three word positions of one slot and nonce are three different permutation outputs + let f: Vec = (0..3).map(|j| scratch_fill(&seed, 12345, 7, 17, j)).collect(); + assert!(f[0] != f[1] && f[1] != f[2] && f[0] != f[2]); +} + +// --------------------------------------------------------------------------------------------------------------- +// 2. Uniformity of the written words and the slot re-hit rate (questions 1 and 2) +// --------------------------------------------------------------------------------------------------------------- + +/// Birthday arithmetic: the expected number of distinct slots after `n` uniform draws from `s` slots. +fn expected_distinct(s: usize, n: usize) -> f64 { + let s = s as f64; + s * (1.0 - (1.0 - 1.0 / s).powi(n as i32)) +} + +struct ClassStats { + units: usize, + events: usize, + hits: usize, + /// ones count per bit of the written words, 3 x 32 + ones: [[u64; 32]; 3], + /// ones count per bit of written XOR read (the change the rewrite makes to the slot) + delta_ones: [[u64; 32]; 3], + /// re-hit depth histogram: how many earlier RMWs the slot had seen in this unit (0 = first touch) + depth: Vec, + max_depth: usize, + /// how often each slot index was addressed (the slot comes from a register's low bits) + slot_hist: Vec, +} + +fn class_stats(name: &str, seeds: &[&str], units_per_seed: usize) -> ClassStats { + let c = class(name); + let mut st = ClassStats { + units: 0, + events: 0, + hits: 0, + ones: [[0; 32]; 3], + delta_ones: [[0; 32]; 3], + depth: vec![0; 256], + max_depth: 0, + slot_hist: vec![0; c.scratch_slots_per_lane()], + }; + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28); + for seed in seeds { + let p = generate_class(seed, c); + assert_eq!(p.scratch_ops_per_hash(), c.scratch_slots() * ITERATIONS); + for u in 0..units_per_seed { + let base = (u as u32).wrapping_mul(32).wrapping_add(0x4000_0000); + let (_, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true); + assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES); + let mut count: HashMap<(u8, u32), usize> = HashMap::new(); + for e in &ev { + assert!(e.slot < c.scratch_slots_per_lane() as u32, "slot inside the lane's scratch"); + let d = count.entry((e.lane, e.slot)).or_insert(0); + assert_eq!(e.hit, *d > 0, "hit flag agrees with the unit's own history"); + assert_eq!(e.written, scratch_rewrite(e.x, &e.read)); + if !e.hit { + let fill = [ + scratch_fill(&p.seed, base, e.lane as u32, e.slot, 0), + scratch_fill(&p.seed, base, e.lane as u32, e.slot, 1), + scratch_fill(&p.seed, base, e.lane as u32, e.slot, 2), + ]; + assert_eq!(e.read, fill, "a first touch reads the fill"); + } + st.depth[(*d).min(255)] += 1; + st.max_depth = st.max_depth.max(*d); + st.slot_hist[e.slot as usize] += 1; + *d += 1; + st.events += 1; + st.hits += e.hit as usize; + for j in 0..3 { + for b in 0..32 { + st.ones[j][b] += ((e.written[j] >> b) & 1) as u64; + st.delta_ones[j][b] += (((e.written[j] ^ e.read[j]) >> b) & 1) as u64; + } + } + } + st.units += 1; + } + } + st +} + +/// Bit bias of every written word (and of the change each rewrite makes) within 6 sigma of a fair coin, over +/// 3 seeds x 2^11 units per class (131,072 hashes per seed set); the slot re-hit rate against the birthday +/// formula within 3 percent relative. The table printed here is the one in the analysis. +#[test] +fn written_words_unbiased_and_rehit_rates() { + let seeds = ["igneum-genesis", "igneum-genesis/stats1", "igneum-genesis/stats2"]; + let units = 1usize << 11; + println!("class | slots/lane | RMW/hash | events | re-hits | re-hit % | birthday % | slot chi2 z (spread) | max depth | max bias sigma | max delta bias sigma"); + for name in CLASSES { + let c = class(name); + let st = class_stats(name, &seeds, units); + let n = st.events as f64; + let sigma = (n / 4.0).sqrt(); + let mut worst = 0.0f64; + let mut worst_delta = 0.0f64; + for j in 0..3 { + for b in 0..32 { + let z = (st.ones[j][b] as f64 - n / 2.0).abs() / sigma; + let zd = (st.delta_ones[j][b] as f64 - n / 2.0).abs() / sigma; + assert!(z <= 6.0, "{name}: written word {j} bit {b} biased: {z:.2} sigma"); + assert!(zd <= 6.0, "{name}: rewrite delta word {j} bit {b} biased: {zd:.2} sigma"); + worst = worst.max(z); + worst_delta = worst_delta.max(zd); + } + } + let per_lane_hash = c.scratch_slots() * ITERATIONS; + let s = c.scratch_slots_per_lane(); + let exp_hits = per_lane_hash as f64 - expected_distinct(s, per_lane_hash); + let exp_pct = 100.0 * exp_hits / per_lane_hash as f64; + let got_pct = 100.0 * st.hits as f64 / st.events as f64; + // chi-square of the slot histogram against uniform (df = s - 1): the slot is a register's low bits, and + // the measured re-hit rate runs above the uniform birthday rate (the finding of the analysis, question 2) + let expect_per_slot = n / s as f64; + let chi2: f64 = st.slot_hist.iter().map(|&h| (h as f64 - expect_per_slot).powi(2) / expect_per_slot).sum(); + let chi2_z = (chi2 - (s as f64 - 1.0)) / (2.0 * (s as f64 - 1.0)).sqrt(); + let hot = *st.slot_hist.iter().max().unwrap() as f64 / expect_per_slot; + let cold = *st.slot_hist.iter().min().unwrap() as f64 / expect_per_slot; + println!( + "{name} | {s} | {per_lane_hash} | {} | {} | {got_pct:.2} | {exp_pct:.2} | {chi2_z:.1} (hottest slot {hot:.2}x, coldest {cold:.2}x) | {} | {worst:.2} | {worst_delta:.2}", + st.events, st.hits, st.max_depth + ); + // a regression band, not a uniformity claim: the rate sits between the uniform birthday rate and twice it + assert!( + got_pct >= 0.9 * exp_pct && got_pct <= 2.0 * exp_pct, + "{name}: re-hit rate {got_pct:.2}% against birthday {exp_pct:.2}%" + ); + // depth histogram: the number of earlier RMWs a re-hit slot had seen in the unit + let shown: Vec = st.depth.iter().take(st.max_depth + 1).enumerate().map(|(d, n)| format!("{d}:{n}")).collect(); + println!(" depth histogram {}", shown.join(" ")); + } +} + +// --------------------------------------------------------------------------------------------------------------- +// 3. Hand-built edge programs against an independent hand model (question 3, CPU half) +// --------------------------------------------------------------------------------------------------------------- + +fn ins(op: Op, dst: u8, src: u8) -> Instr { + Instr { op, dst, src, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1 } +} +fn add_imm(dst: u8, src: u8, imm: u32) -> Instr { + Instr { op: Op::Add, dst, src, src2: 0, imm, imm2: imm, rot: 1, bit: 0, mask: 1, width: 1 } +} + +/// A hand-built program of class `c` named `name` (its seed is the name, so its fill words and init words are +/// its own). These bypass the generator and the acceptance rule, like `TESTS.md` section 2; `sub r, r` zeroes a +/// register as the Swift edge set does. +fn edge(name: &str, c: LoadClass, instrs: Vec) -> Program { + let seed_string = format!("igneum-scratch-edge/{name}"); + let seed_bytes = seed_string.as_bytes().to_vec(); + let k = instrs.iter().filter(|i| i.op == Op::Scratch).count(); + assert_eq!(k, c.scratch_slots(), "{name}: the class carries the program's scratch count"); + Program { + seed: seed_words_from_bytes(&seed_bytes), + seed_string, + seed_bytes, + generator: GENERATOR_VERSION, + attempt: 0, + class: c, + instrs, + } +} + +/// The edge set for a scratch of `kb` KiB per warp. Each entry: (name, what it drives, program). +fn edge_programs(kb: u8) -> Vec<(String, &'static str, Program)> { + let m = LoadClass::scratch(1, kb).scratch_slot_mask(); + let dsts = [2u8, 3, 4, 5, 6, 7, 0, 2, 3, 4, 5, 6, 7, 0, 2, 3]; + let scr = |n: usize, src: u8| -> Vec { (0..n).map(|i| ins(Op::Scratch, dsts[i], src)).collect() }; + let mut v = Vec::new(); + // every RMW of the hash to slot 0 through a zero register: 64 dependent RMWs on one slot per lane + let mut p = vec![ins(Op::Sub, 1, 1)]; + p.extend(scr(8, 1)); + v.push(("slot0".to_string(), "r1 = 0: every RMW to slot 0", edge(&format!("slot0/k{kb}"), LoadClass::scratch(8, kb), p))); + // slot MASK through the in-range register MASK + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m)]; + p.extend(scr(8, 1)); + v.push(("slotmask".to_string(), "r1 = MASK: every RMW to the last slot", edge(&format!("slotmask/k{kb}"), LoadClass::scratch(8, kb), p))); + // slot MASK through the out-of-range register 0xffffffff + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, 1), ins(Op::Sub, 1, 2)]; + p.extend(scr(8, 1)); + v.push(("ones".to_string(), "r1 = 0xffffffff: masked to the last slot", edge(&format!("ones/k{kb}"), LoadClass::scratch(8, kb), p))); + // slot 0 through the out-of-range register MASK + 1 + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m.wrapping_add(1))]; + p.extend(scr(8, 1)); + v.push(("maskplus1".to_string(), "r1 = MASK + 1: masked to slot 0", edge(&format!("maskplus1/k{kb}"), LoadClass::scratch(8, kb), p))); + // 16 RMWs per iteration on slot 0: 128 dependent RMWs on one slot per lane per hash + let mut p = vec![ins(Op::Sub, 1, 1)]; + p.extend(scr(16, 1)); + v.push(("sixteen".to_string(), "16 RMWs per iteration on slot 0", edge(&format!("sixteen/k{kb}"), LoadClass::scratch(16, kb), p))); + // one slot per lane from the init words: lanes with equal slots would show any cross-lane aliasing + // (r5 is the slot register and is never a destination here) + let p: Vec = [0u8, 1, 2, 3, 4, 6, 7, 0].iter().map(|&d| ins(Op::Scratch, d, 5)).collect(); + v.push(("lanevar".to_string(), "r5 never written: one init-dependent slot per lane", edge(&format!("lanevar/k{kb}"), LoadClass::scratch(8, kb), p))); + // alternating slot 0 and slot MASK inside one iteration + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, m)]; + for (i, &d) in [3u8, 4, 5, 6, 7, 0, 3, 4].iter().enumerate() { + // r1 and r2 hold the two slots and are never destinations + p.push(ins(Op::Scratch, d, if i % 2 == 0 { 1 } else { 2 })); + } + v.push(("twoslots".to_string(), "slot 0 and slot MASK alternating", edge(&format!("twoslots/k{kb}"), LoadClass::scratch(8, kb), p))); + v +} + +/// The hand model: a second, minimal interpreter for the ops the edge programs use (sub, add, scratch), with its +/// own slot store keyed by (lane, slot). `mutate` swaps the rewrite's words to show the comparison has teeth. +fn hand_model(p: &Program, base: u32, mutate: bool) -> [u64; 32] { + let seed = &p.seed; + let m = p.class.scratch_slot_mask(); + let mut r = [[0u32; LANES]; 8]; + for lane in 0..LANES { + let nonce = base.wrapping_add(lane as u32); + for i in 0..8 { + let mut x = nonce ^ seed[i]; + x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1)); + x = splitmix32(x); + r[i][lane] = x ^ seed[(i + 1) & 7]; + } + } + let mut store: HashMap<(usize, u32), [u32; 3]> = HashMap::new(); + for _ in 0..ITERATIONS { + let sel = r[0]; + for ins in &p.instrs { + let (d, a) = (ins.dst as usize, ins.src as usize); + match ins.op { + Op::Sub => { + for lane in 0..LANES { + r[d][lane] = r[d][lane].wrapping_sub(r[a][lane]); + } + } + Op::Add => { + for lane in 0..LANES { + let c = if (sel[lane] >> ins.bit) & 1 != 0 { ins.imm2 } else { ins.imm }; + r[d][lane] = r[d][lane].wrapping_add(r[a][lane]).wrapping_add(c); + } + } + Op::Scratch => { + for lane in 0..LANES { + let slot = r[a][lane] & m; + let w = *store.entry((lane, slot)).or_insert_with(|| { + let mut f = [0u32; 3]; + for j in 0..3u32 { + // the fill, written out in full rather than through verify::scratch_fill + let n = base.wrapping_add(lane as u32); + f[j as usize] = splitmix32( + (n ^ seed[j as usize]) + .wrapping_add(slot.wrapping_mul(0x9E37_79B1)) + .wrapping_add((j + 1).wrapping_mul(0x85EB_CA77)), + ); + } + f + }); + let mut x = r[d][lane] ^ w[0]; + x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[1]; + x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[2]; + r[d][lane] = x; + let out = if mutate { + [x.rotate_left(7) ^ w[2], x ^ w[1], x.wrapping_add(w[0])] + } else { + [x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])] + }; + store.insert((lane, slot), out); + } + } + other => panic!("the hand model does not implement {other:?}"), + } + } + } + let mut out = [0u64; 32]; + for lane in 0..LANES { + let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21); + let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27); + out[lane] = ((hi as u64) << 32) | lo as u64; + } + out +} + +/// The four unit bases of every edge vector: 0 and 32 (two consecutive units, the pair a one-warp persistent +/// launch runs on one arena), a unit straddling 2^31, and the unit that wraps past 2^32. +const EDGE_BASES: [u32; 4] = [0, 32, 0x7fff_fff0, 0xffff_ffe0]; + +#[test] +fn edge_programs_match_the_hand_model() { + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24); + let mut cases = 0; + for kb in [32u8, 128] { + for (name, what, p) in edge_programs(kb) { + let slots = p.class.scratch_slots_per_lane(); + for base in EDGE_BASES { + let (res, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true); + let hand = hand_model(&p, base, false); + assert_eq!(res.hashes, hand, "{name} k{kb} base {base:#x}: interpreter against the hand model ({what})"); + assert_ne!(res.hashes, hand_model(&p, base, true), "{name} k{kb}: the comparison has teeth"); + // the slots the trace saw are the ones the program was built to drive + let slot_set: std::collections::BTreeSet = ev.iter().map(|e| e.slot).collect(); + let m = (slots - 1) as u32; + match name.as_str() { + "slot0" | "maskplus1" | "sixteen" => assert_eq!(slot_set.into_iter().collect::>(), vec![0]), + "slotmask" | "ones" => assert_eq!(slot_set.into_iter().collect::>(), vec![m]), + "twoslots" => assert_eq!(slot_set.into_iter().collect::>(), vec![0, m]), + "lanevar" => { + for e in &ev { + assert!(e.slot <= m); + } + } + _ => unreachable!(), + } + // the chain depth on the driven slot: every RMW after the first per lane is a re-hit + let per_lane = p.scratch_ops_per_hash(); + let hits = ev.iter().filter(|e| e.hit).count(); + let expected_hits = match name.as_str() { + "twoslots" => (per_lane - 2) * LANES, + _ => (per_lane - 1) * LANES, + }; + assert_eq!(hits, expected_hits, "{name} k{kb}: re-hits"); + cases += 1; + } + } + } + assert_eq!(cases, 2 * 7 * 4); +} + +// --------------------------------------------------------------------------------------------------------------- +// 4. The static scratch check over every emitted kernel of every scr pack (question 4) +// --------------------------------------------------------------------------------------------------------------- + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Dialect { + Metal, + Cuda, + OpenCl, +} + +/// The static scratch check: every scratch read-modify-write in an emitted kernel has the one masked form the +/// emitter writes, the arena is the lane's own `slots x 4` words, the tag is `salt + unit`, and nothing else +/// touches the scratch. Like the dataset mask check of `TESTS.md` section 5 and `tests/packs.rs`, a text check: +/// the guarantee is that the emitter has one template and it masks. +pub fn scratch_text_check(text: &str, dialect: Dialect, k: usize, slots: usize, kernels: usize) -> Result<(), String> { + assert!(kernels >= 1); + // every count below is per hash kernel; an OpenCL bound file carries igneum_hash and igneum_hash_bound + let k = k * kernels; + assert!(slots.is_power_of_two() && slots >= 1); + let mask = (slots - 1) as u32; + let wpl = slots * 4; + let (u, load, store, ptr) = match dialect { + Dialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "device uint* arena"), + Dialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "uint32_t* arena"), + Dialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); }", "__global uint* arena"), + }; + let count = |needle: &str| text.matches(needle).count(); + let mut errs = Vec::new(); + let mut expect = |what: &str, got: usize, want: usize| { + if got != want { + errs.push(format!("{what}: {got}, expected {want}")); + } + }; + // k slot computations, each masked with exactly the class's mask and immediately followed by the one load form + expect("slot definitions `{ u s_ = r`", count(&format!("{{ {u} s_ = r")), k); + expect("masked slot followed by the load", count(&format!(" & {mask}u; {load}")), k); + expect("stores of the tagged slot", count(store), k); + expect("tag compares", count("(v_.x == tag)"), k); + expect("fill calls (three per RMW)", count("scr_fill(gbase, lane, s_, "), 3 * k); + // the arena: one definition with the class's words per lane, and 2k uses (one load, one store per RMW) + expect("arena definition", count(&format!("{ptr} = scratch + ((size_t)warp_ * 32u + lane) * {wpl}u;")), kernels); + expect("arena mentions (definition + load + store per RMW)", count("arena"), kernels + 2 * k); + expect("tag definition `tag = salt + g_`", count(&format!("{u} tag = salt + g_;")), kernels); + expect("direct scratch indexing", count("scratch["), 0); + expect("scratch pointer arithmetic outside the arena definition", count("scratch +"), kernels); + // no other mask value on a slot: every `s_ = r` line carries the class mask and nothing else carries ` & Nu; uint4 v_` + let any_mask_load = count(&format!("u; {load}")); + expect("loads preceded by some mask (must all be the class mask)", any_mask_load, k); + if errs.is_empty() { + Ok(()) + } else { + Err(errs.join("; ")) + } +} + +fn packs_rw_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-readwidth") +} + +fn scr_packs() -> Vec { + let mut v: Vec = std::fs::read_dir(packs_rw_dir()) + .unwrap() + .map(|d| d.unwrap().file_name().to_string_lossy().to_string()) + .filter(|n| n.starts_with("scr")) + .collect(); + v.sort(); + v +} + +fn read_pack(pack: &str, file: &str) -> String { + let p = packs_rw_dir().join(pack).join(file); + std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display())) +} + +/// Every scr pack regenerates from its program.json (seed bytes, class, day bytes, size) to the same six kernel +/// texts, byte for byte, and every one of those texts passes the static scratch check for the class's k and slot +/// count; the check fails on four deliberate breaks of a copy of the Metal text (mask dropped, mask changed, arena +/// stride changed, a stray scratch access) and on the OpenCL and CUDA twins of the first. +#[test] +fn scr_packs_regenerate_and_pass_the_static_scratch_check() { + let packs = scr_packs(); + assert!(packs.len() >= 6, "the scr packs: {packs:?}"); + let mut checked = 0; + let mut sample_metal = String::new(); + let mut sample_cl = String::new(); + let mut sample_cu = String::new(); + let mut sample_k = 0; + let mut sample_slots = 0; + for pack in &packs { + let j: Value = serde_json::from_str(&read_pack(pack, "program.json")).unwrap(); + let name = j["load_class"].as_str().unwrap(); + let c = class(name); + assert_eq!(&format!("{name}"), pack, "pack directory named after its class"); + let seed = j["seed"].as_str().unwrap(); + let seed_bytes = igneum_pow::bind::unhex(j["seed_bytes"].as_str().unwrap()).unwrap(); + let day_bytes = igneum_pow::bind::unhex(j["dataset"]["day_bytes"].as_str().unwrap()).unwrap(); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard"); + let program = generate_from_seed_bytes_class(seed, &seed_bytes, c); + assert_eq!(program.class, c); + assert_eq!(program.program_id(), u64::from_str_radix(j["program_id"].as_str().unwrap().trim_start_matches("0x"), 16).unwrap()); + let mut dataset = DatasetSource::from_key(seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2); + dataset.key_bytes = day_bytes; + let e = Epoch { program, dataset }; + let p = &e.program; + let mp = e.dataset.memhard().map(|m| &m.params); + let k = c.scratch_slots(); + let slots = c.scratch_slots_per_lane(); + assert_eq!(p.scratch_ops_per_hash(), k * ITERATIONS); + for (file, text, dialect, kernels) in [ + ("program.metal", metal_program(p, log2, LoadSource::Stored), Dialect::Metal, 1), + ("program_bound.metal", metal_program_bound(p, log2), Dialect::Metal, 1), + ("kernel.cu", cuda_kernel(p, mp), Dialect::Cuda, 1), + ("kernel_bound.cu", cuda_kernel_bound(p, mp), Dialect::Cuda, 1), + ("kernel.cl", opencl_kernel(p, mp), Dialect::OpenCl, 1), + // the OpenCL bound file carries igneum_hash and igneum_hash_bound + ("kernel_bound.cl", opencl_kernel_bound(p, mp), Dialect::OpenCl, 2), + ] { + let on_disk = read_pack(pack, file); + assert_eq!(on_disk, text, "{pack}/{file}: the pack is the emitter's text"); + // scr0 is the persistent control: an arena and a tag, no read-modify-write; the check holds with k = 0 + scratch_text_check(&on_disk, dialect, k, slots, kernels).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")); + checked += 1; + } + // the vectors of the pack are the CPU's + let v: Value = serde_json::from_str(&read_pack(pack, "vectors.json")).unwrap(); + for w in v["warps"].as_array().unwrap() { + let base = w["base_nonce"].as_u64().unwrap() as u32; + let got = e.hash_warp(base); + for (lane, x) in w["expected"].as_array().unwrap().iter().enumerate() { + let want = u64::from_str_radix(x.as_str().unwrap().trim_start_matches("0x"), 16).unwrap(); + assert_eq!(got[lane], want, "{pack}: base {base} lane {lane}"); + } + } + if k == 4 && slots == 64 { + sample_metal = read_pack(pack, "program.metal"); + sample_cl = read_pack(pack, "kernel.cl"); + sample_cu = read_pack(pack, "kernel.cu"); + sample_k = k; + sample_slots = slots; + } + } + assert_eq!(checked, packs.len() * 6); + println!("static scratch check: {checked} kernels over {} scr packs", packs.len()); + + // The deliberate breaks (the watcher rule of CLAUDE.md: a check is trusted once it fails on a known-broken + // case). Each must be caught; the message names what. + assert!(sample_k == 4 && sample_slots == 64, "scr4k32 is in the pack set"); + let mask = format!(" & {}u; uint4 v_", sample_slots - 1); + let broken_mask = sample_metal.replacen(&mask, "; uint4 v_", 1); + assert_ne!(broken_mask, sample_metal); + let e = scratch_text_check(&broken_mask, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); + println!("break 1 (one mask dropped, Metal): {e}"); + let wrong_mask = sample_metal.replace(" & 63u;", " & 127u;"); + let e = scratch_text_check(&wrong_mask, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 0, expected 4"), "{e}"); + println!("break 2 (mask 63 -> 127 on every RMW, Metal): {e}"); + let wrong_stride = sample_metal.replace("* 256u;", "* 128u;"); + let e = scratch_text_check(&wrong_stride, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("arena definition: 0, expected 1"), "{e}"); + println!("break 3 (arena stride 256 -> 128 words, Metal): {e}"); + let stray = format!("{sample_metal}\n// stray\n// arena[0] = 0u; scratch[1] = 1u;\n"); + let e = scratch_text_check(&stray, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("arena mentions") && e.contains("direct scratch indexing: 1, expected 0"), "{e}"); + println!("break 4 (a stray arena and scratch access, Metal): {e}"); + let e = scratch_text_check(&sample_cl.replacen(" & 63u; uint4 v_ = vload4", "; uint4 v_ = vload4", 1), Dialect::OpenCl, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); + println!("break 5 (one mask dropped, OpenCL): {e}"); + let e = scratch_text_check(&sample_cu.replacen(" & 63u; uint4 v_ = *(const uint4*)", "; uint4 v_ = *(const uint4*)", 1), Dialect::Cuda, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); + println!("break 6 (one mask dropped, CUDA): {e}"); + // and the unbroken texts pass under the same calls + scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 1).unwrap(); + scratch_text_check(&sample_cl, Dialect::OpenCl, 4, 64, 1).unwrap(); + scratch_text_check(&sample_cu, Dialect::Cuda, 4, 64, 1).unwrap(); + // a wrong slot count, RMW count or kernel count against a right text fails too (the check is tied to the class) + assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 256, 1).is_err()); + assert!(scratch_text_check(&sample_metal, Dialect::Metal, 3, 64, 1).is_err()); + assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 2).is_err()); +} + +// --------------------------------------------------------------------------------------------------------------- +// 5. The fuzz: 200 generated scratch programs, contract on every instruction, 4 units each across the 32-bit +// range including the wrap; with IGNEUM_SCRATCH_PACKS_OUT the packs for the Metal runs (question 3, 4) +// --------------------------------------------------------------------------------------------------------------- + +/// Write a pack whose vectors.json carries `bases` (any number of units) instead of the three standard bases. +fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) -> Vec<[u64; 32]> { + let mut pack = export_pack(e, day, source); + let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect(); + let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true); + for f in pack.files.iter_mut() { + if f.0 == "vectors.json" { + f.1 = vj.clone(); + } + } + pack.write_to(dir).unwrap(); + outs +} + +fn contract(p: &Program) { + assert_eq!(p.instrs.len(), INSTR_COUNT); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count() + p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), 16); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), p.class.scratch_slots()); + assert!(p.instrs[0].op != Op::Load && p.instrs[0].op != Op::Scratch, "instruction 0 is never a memory op"); + for (k, i) in p.instrs.iter().enumerate() { + assert!(i.src != i.dst, "#{k}: src == dst"); + assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot); + assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask); + assert!(i.dst < 8 && i.src < 8 && i.src2 < 8); + assert_eq!(i.width, 1, "#{k}: a scratch class reads one-word loads"); + } + assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program"); +} + +#[test] +fn fuzz_scr_programs_cpu() { + let n: usize = std::env::var("IGNEUM_SCRATCH_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200); + let out = std::env::var("IGNEUM_SCRATCH_PACKS_OUT").ok().map(PathBuf::from); + let mut rng = SplitMix64::new(0x6967_6e65_756d_2d73); // "igneum-s" + let day = "2026-10-03"; + let closed = DatasetSource::new(day, DatasetMode::ClosedForm, 28); + // memory-hard sources per size, built once each (the cache fill is 0.2 s); only when packs are written + let mut mh: HashMap = HashMap::new(); + let mut manifest = String::from("pack\tclass\tlog2\tprogram_id\tscratch_ops_per_hash\tbases\n"); + let mut per_class: HashMap = HashMap::new(); + let mut units = 0usize; + let mut wraps = 0usize; + if let Some(dir) = &out { + std::fs::create_dir_all(dir).unwrap(); + // the edge packs first: 64 MiB datasets (no dataset load in them), the four edge bases + for kb in [32u8, 128] { + for (name, _what, p) in edge_programs(kb) { + let log2 = 24; + let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2)); + let e = Epoch { program: p, dataset: ds }; + let pack_name = format!("edge-{name}-k{kb}"); + write_pack_with_bases(&dir.join(&pack_name), &e, day, &EDGE_BASES, "igneum-pow tests/scratch.rs edge"); + manifest.push_str(&format!( + "{pack_name}\t{}\t{log2}\t{:016x}\t{}\t{}\n", + e.program.class.name(), + e.program.program_id(), + e.program.scratch_ops_per_hash(), + EDGE_BASES.iter().map(|b| format!("{b}")).collect::>().join(",") + )); + mh.insert(log2, e.dataset); + } + } + } + for i in 0..n { + let name = CLASSES[rng.below(CLASSES.len() as u64) as usize]; + let c = class(name); + let seed = format!("igneum-scratch-fuzz/{i}"); + let p = generate_class(&seed, c); + contract(&p); + *per_class.entry(name.to_string()).or_insert(0) += 1; + // four bases: one inside a 256-nonce batch (in-batch check on the GPU), one straddling 2^31, one in + // the last 256 nonces (the unit wraps past 2^32 or ends on it), one uniform + let b0 = (rng.below(8) as u32) * 32; + let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32); + let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32); + let b3 = (rng.next() as u32) & !31; + let bases = [b0, b1, b2, b3]; + // an aligned unit never straddles 2^32 (spec 1.9); the top unit ends on 0xffffffff and the persistent + // kernel's unit sequence wraps inside a launch, which the Metal run checks with packbench --batch-base + wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count(); + // the CPU: the interpreter is deterministic and every scratch event is inside the lane's slots + for &b in &bases { + let (r1, ev) = interpret_warp_scratch(&p, &p.seed, b, &closed, true); + let r2 = interpret_warp_scratch(&p, &p.seed, b, &closed, false).0; + assert_eq!(r1.hashes, r2.hashes); + assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES); + assert!(ev.iter().all(|e: &ScratchEvent| e.slot < c.scratch_slots_per_lane() as u32)); + units += 1; + } + if let Some(dir) = &out { + let log2 = [24u32, 26, 28][rng.below(3) as usize]; + let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2)); + let e = Epoch { program: p, dataset: ds }; + let pack_name = format!("fuzz-{i:03}-{name}-l{log2}"); + write_pack_with_bases(&dir.join(&pack_name), &e, day, &bases, "igneum-pow tests/scratch.rs fuzz"); + manifest.push_str(&format!( + "{pack_name}\t{name}\t{log2}\t{:016x}\t{}\t{}\n", + e.program.program_id(), + e.program.scratch_ops_per_hash(), + bases.iter().map(|b| format!("{b}")).collect::>().join(",") + )); + mh.insert(log2, e.dataset); + } else { + let _ = rng.below(3); + } + } + let mut classes: Vec<_> = per_class.iter().collect(); + classes.sort(); + println!("fuzz: {n} programs, {units} units on the CPU, {wraps} units in the top 256 nonces, classes {classes:?}"); + assert_eq!(units, 4 * n); + assert_eq!(wraps, n, "every program has a unit in the top 256 nonces"); + if let Some(dir) = &out { + std::fs::write(dir.join("manifest.tsv"), manifest).unwrap(); + println!("packs written to {}", dir.display()); + } +} + +/// The fold and rewrite, restated: a slot after `d` dependent RMWs holds 96 bits that are a function of the fill +/// (3 words, a pure function of nonce, slot and seed) and the `d` fold values; a chip that keeps the `d` fold +/// values (32 bits each) instead of the 96-bit slot recomputes the slot in `d` rewrites. This test pins the +/// arithmetic the analysis uses (question 2): the replay from the fold values reproduces the slot. +#[test] +fn slot_is_replayable_from_its_fold_values() { + let seed = seed_words_from_bytes(b"igneum-genesis"); + let (base, lane, slot) = (0x1234_5600u32, 5u32, 17u32); + let fill = [scratch_fill(&seed, base, lane, slot, 0), scratch_fill(&seed, base, lane, slot, 1), scratch_fill(&seed, base, lane, slot, 2)]; + let mut rng = SplitMix64::new(99); + let dsts: Vec = (0..64).map(|_| rng.next() as u32).collect(); + // the honest sequence: read, fold, rewrite, 64 times + let mut w = fill; + let mut xs = Vec::new(); + for &d in &dsts { + let x = fold_words(d, &w); + xs.push(x); + w = scratch_rewrite(x, &w); + } + // the replay: from the fill and the stored fold values alone + let mut w2 = fill; + for &x in &xs { + w2 = scratch_rewrite(x, &w2); + } + assert_eq!(w, w2); + // and nothing shorter: the fold value at step d depends on the slot content at step d, which depends on + // every earlier fold value (drop one and the chain diverges) + let mut w3 = fill; + for (i, &x) in xs.iter().enumerate() { + if i != 10 { + w3 = scratch_rewrite(x, &w3); + } + } + assert_ne!(w, w3); +} diff --git a/proto-metal/packbench.swift b/proto-metal/packbench.swift index 75b6fc4f..dfa07e44 100644 --- a/proto-metal/packbench.swift +++ b/proto-metal/packbench.swift @@ -5,7 +5,7 @@ // against the Rust CPU reference and timed without a Swift mirror of the generator. One file, no packages. // // swiftc -O -target arm64-apple-macos11 -o packbench packbench.swift -framework Metal -// ./packbench --pack [--batches 5] [--batch-log2 24] [--group 256] [--warps 2048] +// ./packbench --pack [--batches 5] [--batch-log2 24] [--group 256] [--warps 2048] [--batch-base 0] // // Prints one RESULT line per run: vectors, cache and dataset checks, the batch fingerprint (FNV-1a 64 over the 2^B // outputs at base nonce 0) and MH/s by wall and by GPU time. Variant 5 packs (IGNEUM_PERSISTENT_WARPS) are launched @@ -16,7 +16,7 @@ import Metal func nowMs() -> Double { return Double(DispatchTime.now().uptimeNanoseconds) / 1e6 } func fail(_ m: String) -> Never { print("FAIL: \(m)"); exit(1) } -struct Opts { var pack = ""; var batches = 5; var batchLog2 = 24; var group = 256; var warps = 2048 } +struct Opts { var pack = ""; var batches = 5; var batchLog2 = 24; var group = 256; var warps = 2048; var batchBase: UInt32 = 0 } var opts = Opts() var args = Array(CommandLine.arguments.dropFirst()) while !args.isEmpty { @@ -28,6 +28,7 @@ while !args.isEmpty { case "--batch-log2": opts.batchLog2 = Int(next())! case "--group": opts.group = Int(next())! case "--warps": opts.warps = Int(next())! + case "--batch-base": opts.batchBase = UInt32(next())! // base nonce of the fingerprint batch (default 0; a base near 2^32 makes the persistent unit sequence wrap inside the launch) default: fail("unknown argument \(a)") } } @@ -183,12 +184,13 @@ for (i, base) in vecBases.enumerated() { var nonces = 1 << opts.batchLog2 if persistent { let unit = 32 * min(warpsN, nonces / 32); nonces = (nonces / unit) * unit } let out = device.makeBuffer(length: nonces * 8, options: .storageModeShared)! -let (warmWall, warmGpu) = run { enc in encodeHash(enc, out: out, base: 0, nonces: nonces, group: opts.group) } +let (warmWall, warmGpu) = run { enc in encodeHash(enc, out: out, base: opts.batchBase, nonces: nonces, group: opts.group) } let outPtr = out.contents().bindMemory(to: UInt64.self, capacity: nonces) var batchVecPass = 0, batchVecN = 0 -for (i, base) in vecBases.enumerated() where Int(base) + 32 <= nonces { +for (i, base) in vecBases.enumerated() where Int(base &- opts.batchBase) + 32 <= nonces { // the batch window, wrapping past 2^32 + let off = Int(base &- opts.batchBase) batchVecN += 1 - if (0..<32).allSatisfy({ outPtr[Int(base) + $0] == vecOuts[i][$0] }) { batchVecPass += 1 } + if (0..<32).allSatisfy({ outPtr[off + $0] == vecOuts[i][$0] }) { batchVecPass += 1 } } let fingerprint = fnv1a64(out.contents(), nonces * 8) // Timed batches @@ -205,5 +207,5 @@ print("device \(device.name); compile \(String(format: "%.0f", compileMs)) ms; c print("cache FNV-1a 64 \(String(format: "%016llx", cacheFnv)) \(cacheOk ? "PASS" : "FAIL"); dataset head and last \(dsOk ? "PASS" : "FAIL"); vectors standalone \(vecPass)/\(vecBases.count), in batch \(batchVecPass)/\(batchVecN)") print("warm-up batch \(nonces) hashes: \(String(format: "%.1f", warmGpu)) ms GPU, \(String(format: "%.1f", warmWall)) ms wall") let overall = cacheOk && dsOk && vecPass == vecBases.count && batchVecPass == batchVecN -print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) arena_mib=\(persistent ? warpsN : 0) nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")") +print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) batch_base=\(opts.batchBase) arena_mib=\(persistent ? warpsN : 0) nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")") exit(overall ? 0 : 1)