scratch soundness (layer 3 of Counter ASIC 2.0): verify.rs scratch trace hook; tests/scratch.rs: rewrite and fill bijections, written-word bias and re-hit rates per class, hand-built slot edge programs against a hand model, static scratch-mask check over every emitted kernel of every scr pack with six deliberate breaks, 200-program CPU fuzz and pack writer for the Metal runs; packbench --batch-base for the launch-level nonce wrap
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
ea863d6d4b
commit
c4ce51a14f
3 changed files with 815 additions and 8 deletions
|
|
@ -51,6 +51,22 @@ pub struct ScratchModel {
|
|||
data: Vec<[u32; 3]>,
|
||||
pub reads: usize,
|
||||
pub writes: usize,
|
||||
/// Soundness tests (`tests/scratch.rs`, `docs/analysis/scratch-soundness.md`): when `Some`, every
|
||||
/// read-modify-write is appended as it happened. `None` on every verification path.
|
||||
pub trace: Option<Vec<ScratchEvent>>,
|
||||
}
|
||||
|
||||
/// One scratch read-modify-write as the interpreter saw it (variant 5 soundness tests).
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub struct ScratchEvent {
|
||||
pub lane: u8,
|
||||
pub slot: u32,
|
||||
/// The slot had been written earlier in this unit (a re-hit): the words read were a rewrite, not the fill.
|
||||
pub hit: bool,
|
||||
pub read: [u32; 3],
|
||||
/// The fold result, the new value of `dst`.
|
||||
pub x: u32,
|
||||
pub written: [u32; 3],
|
||||
}
|
||||
|
||||
impl ScratchModel {
|
||||
|
|
@ -61,6 +77,7 @@ impl ScratchModel {
|
|||
data: vec![[0; 3]; LANES * slots_per_lane],
|
||||
reads: 0,
|
||||
writes: 0,
|
||||
trace: None,
|
||||
}
|
||||
}
|
||||
/// Read slot `slot` of `lane`, then rewrite it from the fold result `x`. Returns the three words read.
|
||||
|
|
@ -77,7 +94,11 @@ impl ScratchModel {
|
|||
]
|
||||
};
|
||||
let x = fold_words(dst, &w);
|
||||
self.data[i] = scratch_rewrite(x, &w);
|
||||
let out = scratch_rewrite(x, &w);
|
||||
if let Some(t) = self.trace.as_mut() {
|
||||
t.push(ScratchEvent { lane: lane as u8, slot, hit: self.written[i], read: w, x, written: out });
|
||||
}
|
||||
self.data[i] = out;
|
||||
self.written[i] = true;
|
||||
self.reads += 1;
|
||||
self.writes += 1;
|
||||
|
|
@ -243,6 +264,19 @@ pub fn interpret_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) ->
|
|||
/// [`interpret_warp`] with explicit init words `I` (section 1.6 of the spec). The packs use `I = program.seed`;
|
||||
/// a block uses `I = bind::block_init_words(H, nonce)`.
|
||||
pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, ds: &DatasetSource) -> WarpResult {
|
||||
interpret_warp_scratch(program, seed, base_nonce, ds, false).0
|
||||
}
|
||||
|
||||
/// [`interpret_warp_init`] that also returns every scratch read-modify-write of the unit in execution order
|
||||
/// (lane-minor within an instruction, as the interpreter runs them) when `trace` is set; empty otherwise and for
|
||||
/// a class without a scratch. For the soundness tests of variant 5 only.
|
||||
pub fn interpret_warp_scratch(
|
||||
program: &Program,
|
||||
seed: &[u32; 8],
|
||||
base_nonce: u32,
|
||||
ds: &DatasetSource,
|
||||
trace: bool,
|
||||
) -> (WarpResult, Vec<ScratchEvent>) {
|
||||
let mask = ds.mask;
|
||||
let mut r = [[0u32; LANES]; 8];
|
||||
for lane in 0..LANES {
|
||||
|
|
@ -258,6 +292,11 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32,
|
|||
let mut idx = [0u32; LANES];
|
||||
let mut val = [0u32; LANES];
|
||||
let mut scratch = if program.has_scratch() { Some(ScratchModel::new(program.class.scratch_slots_per_lane())) } else { None };
|
||||
if trace {
|
||||
if let Some(m) = scratch.as_mut() {
|
||||
m.trace = Some(Vec::new());
|
||||
}
|
||||
}
|
||||
let slot_mask = program.class.scratch_slot_mask();
|
||||
for _ in 0..ITERATIONS {
|
||||
let sel = r[0];
|
||||
|
|
@ -279,7 +318,8 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32,
|
|||
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
|
||||
hashes[lane] = ((hi as u64) << 32) | lo as u64;
|
||||
}
|
||||
WarpResult { hashes, items_derived }
|
||||
let events = scratch.and_then(|m| m.trace).unwrap_or_default();
|
||||
(WarpResult { hashes, items_derived }, events)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
|
|
|
|||
765
igneum-pow/tests/scratch.rs
Normal file
765
igneum-pow/tests/scratch.rs
Normal file
|
|
@ -0,0 +1,765 @@
|
|||
//! Soundness tests of layer 3 of `docs/plans/counter-asic-2.md`: the per-warp scratch with read-modify-writes
|
||||
//! (variant 5 of the read-width experiment, `LoadClass::scratch(k, kb)`). Analysis and results:
|
||||
//! `docs/analysis/scratch-soundness.md`. Every test is parametric over the class's slot count
|
||||
//! (`scratch_slots_per_lane()`), so the 32 and 128 KiB geometries and any later one run the same checks.
|
||||
//!
|
||||
//! What runs under plain `cargo test`:
|
||||
//! 1. `rewrite_is_a_bijection_of_the_fold_value`, `fill_is_a_bijection_of_the_nonce`: the written words as
|
||||
//! functions (question 1).
|
||||
//! 2. `written_words_unbiased_and_rehit_rates`: bit bias of every written word over 2^11 units x 3 seeds per class
|
||||
//! (the TESTS.md section 3 shape), and the measured slot re-hit rate against the birthday formula (question 2).
|
||||
//! 3. `edge_programs_match_the_hand_model`: hand-built programs that drive every read-modify-write of a hash to
|
||||
//! slot 0, slot MASK, through out-of-range registers, to one slot per lane, alternating two slots, and 16
|
||||
//! read-modify-writes per iteration on one slot; the interpreter against an independent hand model, and the
|
||||
//! hand model shown to have teeth (question 3, CPU half).
|
||||
//! 4. `scr_packs_regenerate_and_pass_the_static_scratch_check`: every emitted kernel of every scr pack under
|
||||
//! `proto-cuda/packs-readwidth` regenerates from its program.json and passes the static scratch-mask check;
|
||||
//! the check is shown to fail on four deliberate breaks (question 4).
|
||||
//! 5. `fuzz_scr_programs_cpu`: 200 generated scratch programs over the six classes, generator contract on every
|
||||
//! instruction, 4 units each at base nonces across the 32-bit range including the wrap; with
|
||||
//! `IGNEUM_SCRATCH_PACKS_OUT=<dir>` it also writes the packs (and the edge packs) for the Metal runs of
|
||||
//! `proto-metal/packbench` (question 3 GPU half, question 4, `TESTS.md` section 9 shape).
|
||||
|
||||
use igneum_pow::emit::{
|
||||
cuda_kernel, cuda_kernel_bound, export_pack, metal_program, metal_program_bound, opencl_kernel,
|
||||
opencl_kernel_bound, vectors_json, LoadSource,
|
||||
};
|
||||
use igneum_pow::generator::{
|
||||
generate_class, generate_from_seed_bytes_class, Instr, LoadClass, Op, Program, GENERATOR_VERSION, INSTR_COUNT,
|
||||
ITERATIONS, LANES,
|
||||
};
|
||||
use igneum_pow::seed::{seed_words_from_bytes, SplitMix64};
|
||||
use igneum_pow::verify::{
|
||||
fold_words, interpret_warp_scratch, scratch_fill, scratch_rewrite, splitmix32, DatasetMode, DatasetSource,
|
||||
Epoch, ScratchEvent, FOLD_MUL, FOLD_ROT,
|
||||
};
|
||||
use serde_json::Value;
|
||||
use std::collections::HashMap;
|
||||
use std::path::PathBuf;
|
||||
|
||||
/// The classes under study: the two capped geometries (32 and 128 KiB per warp: 64 and 256 slots per lane) at the
|
||||
/// RMW shares the readwidth branch measures.
|
||||
const CLASSES: [&str; 6] = ["scr2k32", "scr4k32", "scr8k32", "scr2k128", "scr4k128", "scr8k128"];
|
||||
|
||||
fn class(name: &str) -> LoadClass {
|
||||
LoadClass::parse(name).unwrap_or_else(|| panic!("class {name}"))
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
// 1. The written words as functions (question 1)
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
|
||||
/// For a fixed slot content `w`, each of the three rewritten words is a bijection of the fold value `x`
|
||||
/// (`x ^ w1`, `rotl(x, 7) ^ w2`, `x + w0`), so the rewrite is injective in `x` and a uniform `x` gives a uniform
|
||||
/// word in every position. Checked over 2^16 consecutive `x` for 16 random `w`.
|
||||
#[test]
|
||||
fn rewrite_is_a_bijection_of_the_fold_value() {
|
||||
let mut rng = SplitMix64::new(0x7363_7261_7463_6801);
|
||||
for _ in 0..16 {
|
||||
let w = [rng.next() as u32, rng.next() as u32, rng.next() as u32];
|
||||
let x0 = rng.next() as u32;
|
||||
let mut seen = [vec![false; 1 << 16], vec![false; 1 << 16], vec![false; 1 << 16]];
|
||||
for i in 0..(1u32 << 16) {
|
||||
let x = x0.wrapping_add(i);
|
||||
let out = scratch_rewrite(x, &w);
|
||||
for j in 0..3 {
|
||||
// a bijection of x maps 2^16 consecutive x to 2^16 distinct words; the low 16 bits alone are
|
||||
// distinct for the xor words (x ^ c) and for the add word (x + c), since both act on the low 16
|
||||
// bits as bijections of the low 16 bits of x; the rotl word is checked on its rotated-back bits
|
||||
let key = if j == 1 { out[j].rotate_right(7) & 0xffff } else { out[j] & 0xffff };
|
||||
assert!(!seen[j][key as usize], "word {j} repeats inside 2^16 consecutive x");
|
||||
seen[j][key as usize] = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
// The rewrite inverts: from the old content and any ONE written word the fold value is recovered, so a
|
||||
// rewritten slot carries exactly 32 bits of new state (the point of question 2's arithmetic).
|
||||
let w = [0x1234_5678, 0x9abc_def0, 0x0fed_cba9];
|
||||
let x = 0xdead_beef;
|
||||
let out = scratch_rewrite(x, &w);
|
||||
assert_eq!(out[0] ^ w[1], x);
|
||||
assert_eq!((out[1] ^ w[2]).rotate_right(7), x);
|
||||
assert_eq!(out[2].wrapping_sub(w[0]), x);
|
||||
}
|
||||
|
||||
/// For a fixed (seed, slot, j) the fill is a bijection of the lane nonce: `splitmix32` is a bijection of its
|
||||
/// 32-bit input and the input `((base + lane) ^ s) + c` is a bijection of `base + lane`. Over 2^16 consecutive
|
||||
/// nonces no fill word repeats, for 8 slots x 3 words.
|
||||
#[test]
|
||||
fn fill_is_a_bijection_of_the_nonce() {
|
||||
let seed = seed_words_from_bytes(b"igneum-genesis");
|
||||
for slot in [0u32, 1, 63, 64, 255, 1023, 2047] {
|
||||
for j in 0..3u32 {
|
||||
let mut words: Vec<u32> = (0..(1u32 << 16)).map(|n| scratch_fill(&seed, n, 0, slot, j)).collect();
|
||||
words.sort_unstable();
|
||||
words.dedup();
|
||||
assert_eq!(words.len(), 1 << 16, "slot {slot} word {j}: fill words of 2^16 consecutive nonces are distinct");
|
||||
}
|
||||
}
|
||||
// base + lane is the lane nonce: the fill of lane l at base b is the fill of lane 0 at base b + l
|
||||
assert_eq!(scratch_fill(&seed, 0x1000, 7, 5, 2), scratch_fill(&seed, 0x1007, 0, 5, 2));
|
||||
// and it wraps with the nonce: base 0xffffffe0, lane 31 is nonce 0xffffffff; lane 32 would be nonce 0
|
||||
assert_eq!(scratch_fill(&seed, 0xffff_ffe0, 32, 5, 2), scratch_fill(&seed, 0, 0, 5, 2));
|
||||
// the three word positions of one slot and nonce are three different permutation outputs
|
||||
let f: Vec<u32> = (0..3).map(|j| scratch_fill(&seed, 12345, 7, 17, j)).collect();
|
||||
assert!(f[0] != f[1] && f[1] != f[2] && f[0] != f[2]);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
// 2. Uniformity of the written words and the slot re-hit rate (questions 1 and 2)
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
|
||||
/// Birthday arithmetic: the expected number of distinct slots after `n` uniform draws from `s` slots.
|
||||
fn expected_distinct(s: usize, n: usize) -> f64 {
|
||||
let s = s as f64;
|
||||
s * (1.0 - (1.0 - 1.0 / s).powi(n as i32))
|
||||
}
|
||||
|
||||
struct ClassStats {
|
||||
units: usize,
|
||||
events: usize,
|
||||
hits: usize,
|
||||
/// ones count per bit of the written words, 3 x 32
|
||||
ones: [[u64; 32]; 3],
|
||||
/// ones count per bit of written XOR read (the change the rewrite makes to the slot)
|
||||
delta_ones: [[u64; 32]; 3],
|
||||
/// re-hit depth histogram: how many earlier RMWs the slot had seen in this unit (0 = first touch)
|
||||
depth: Vec<usize>,
|
||||
max_depth: usize,
|
||||
/// how often each slot index was addressed (the slot comes from a register's low bits)
|
||||
slot_hist: Vec<u64>,
|
||||
}
|
||||
|
||||
fn class_stats(name: &str, seeds: &[&str], units_per_seed: usize) -> ClassStats {
|
||||
let c = class(name);
|
||||
let mut st = ClassStats {
|
||||
units: 0,
|
||||
events: 0,
|
||||
hits: 0,
|
||||
ones: [[0; 32]; 3],
|
||||
delta_ones: [[0; 32]; 3],
|
||||
depth: vec![0; 256],
|
||||
max_depth: 0,
|
||||
slot_hist: vec![0; c.scratch_slots_per_lane()],
|
||||
};
|
||||
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28);
|
||||
for seed in seeds {
|
||||
let p = generate_class(seed, c);
|
||||
assert_eq!(p.scratch_ops_per_hash(), c.scratch_slots() * ITERATIONS);
|
||||
for u in 0..units_per_seed {
|
||||
let base = (u as u32).wrapping_mul(32).wrapping_add(0x4000_0000);
|
||||
let (_, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true);
|
||||
assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES);
|
||||
let mut count: HashMap<(u8, u32), usize> = HashMap::new();
|
||||
for e in &ev {
|
||||
assert!(e.slot < c.scratch_slots_per_lane() as u32, "slot inside the lane's scratch");
|
||||
let d = count.entry((e.lane, e.slot)).or_insert(0);
|
||||
assert_eq!(e.hit, *d > 0, "hit flag agrees with the unit's own history");
|
||||
assert_eq!(e.written, scratch_rewrite(e.x, &e.read));
|
||||
if !e.hit {
|
||||
let fill = [
|
||||
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 0),
|
||||
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 1),
|
||||
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 2),
|
||||
];
|
||||
assert_eq!(e.read, fill, "a first touch reads the fill");
|
||||
}
|
||||
st.depth[(*d).min(255)] += 1;
|
||||
st.max_depth = st.max_depth.max(*d);
|
||||
st.slot_hist[e.slot as usize] += 1;
|
||||
*d += 1;
|
||||
st.events += 1;
|
||||
st.hits += e.hit as usize;
|
||||
for j in 0..3 {
|
||||
for b in 0..32 {
|
||||
st.ones[j][b] += ((e.written[j] >> b) & 1) as u64;
|
||||
st.delta_ones[j][b] += (((e.written[j] ^ e.read[j]) >> b) & 1) as u64;
|
||||
}
|
||||
}
|
||||
}
|
||||
st.units += 1;
|
||||
}
|
||||
}
|
||||
st
|
||||
}
|
||||
|
||||
/// Bit bias of every written word (and of the change each rewrite makes) within 6 sigma of a fair coin, over
|
||||
/// 3 seeds x 2^11 units per class (131,072 hashes per seed set); the slot re-hit rate against the birthday
|
||||
/// formula within 3 percent relative. The table printed here is the one in the analysis.
|
||||
#[test]
|
||||
fn written_words_unbiased_and_rehit_rates() {
|
||||
let seeds = ["igneum-genesis", "igneum-genesis/stats1", "igneum-genesis/stats2"];
|
||||
let units = 1usize << 11;
|
||||
println!("class | slots/lane | RMW/hash | events | re-hits | re-hit % | birthday % | slot chi2 z (spread) | max depth | max bias sigma | max delta bias sigma");
|
||||
for name in CLASSES {
|
||||
let c = class(name);
|
||||
let st = class_stats(name, &seeds, units);
|
||||
let n = st.events as f64;
|
||||
let sigma = (n / 4.0).sqrt();
|
||||
let mut worst = 0.0f64;
|
||||
let mut worst_delta = 0.0f64;
|
||||
for j in 0..3 {
|
||||
for b in 0..32 {
|
||||
let z = (st.ones[j][b] as f64 - n / 2.0).abs() / sigma;
|
||||
let zd = (st.delta_ones[j][b] as f64 - n / 2.0).abs() / sigma;
|
||||
assert!(z <= 6.0, "{name}: written word {j} bit {b} biased: {z:.2} sigma");
|
||||
assert!(zd <= 6.0, "{name}: rewrite delta word {j} bit {b} biased: {zd:.2} sigma");
|
||||
worst = worst.max(z);
|
||||
worst_delta = worst_delta.max(zd);
|
||||
}
|
||||
}
|
||||
let per_lane_hash = c.scratch_slots() * ITERATIONS;
|
||||
let s = c.scratch_slots_per_lane();
|
||||
let exp_hits = per_lane_hash as f64 - expected_distinct(s, per_lane_hash);
|
||||
let exp_pct = 100.0 * exp_hits / per_lane_hash as f64;
|
||||
let got_pct = 100.0 * st.hits as f64 / st.events as f64;
|
||||
// chi-square of the slot histogram against uniform (df = s - 1): the slot is a register's low bits, and
|
||||
// the measured re-hit rate runs above the uniform birthday rate (the finding of the analysis, question 2)
|
||||
let expect_per_slot = n / s as f64;
|
||||
let chi2: f64 = st.slot_hist.iter().map(|&h| (h as f64 - expect_per_slot).powi(2) / expect_per_slot).sum();
|
||||
let chi2_z = (chi2 - (s as f64 - 1.0)) / (2.0 * (s as f64 - 1.0)).sqrt();
|
||||
let hot = *st.slot_hist.iter().max().unwrap() as f64 / expect_per_slot;
|
||||
let cold = *st.slot_hist.iter().min().unwrap() as f64 / expect_per_slot;
|
||||
println!(
|
||||
"{name} | {s} | {per_lane_hash} | {} | {} | {got_pct:.2} | {exp_pct:.2} | {chi2_z:.1} (hottest slot {hot:.2}x, coldest {cold:.2}x) | {} | {worst:.2} | {worst_delta:.2}",
|
||||
st.events, st.hits, st.max_depth
|
||||
);
|
||||
// a regression band, not a uniformity claim: the rate sits between the uniform birthday rate and twice it
|
||||
assert!(
|
||||
got_pct >= 0.9 * exp_pct && got_pct <= 2.0 * exp_pct,
|
||||
"{name}: re-hit rate {got_pct:.2}% against birthday {exp_pct:.2}%"
|
||||
);
|
||||
// depth histogram: the number of earlier RMWs a re-hit slot had seen in the unit
|
||||
let shown: Vec<String> = st.depth.iter().take(st.max_depth + 1).enumerate().map(|(d, n)| format!("{d}:{n}")).collect();
|
||||
println!(" depth histogram {}", shown.join(" "));
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
// 3. Hand-built edge programs against an independent hand model (question 3, CPU half)
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
|
||||
fn ins(op: Op, dst: u8, src: u8) -> Instr {
|
||||
Instr { op, dst, src, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1 }
|
||||
}
|
||||
fn add_imm(dst: u8, src: u8, imm: u32) -> Instr {
|
||||
Instr { op: Op::Add, dst, src, src2: 0, imm, imm2: imm, rot: 1, bit: 0, mask: 1, width: 1 }
|
||||
}
|
||||
|
||||
/// A hand-built program of class `c` named `name` (its seed is the name, so its fill words and init words are
|
||||
/// its own). These bypass the generator and the acceptance rule, like `TESTS.md` section 2; `sub r, r` zeroes a
|
||||
/// register as the Swift edge set does.
|
||||
fn edge(name: &str, c: LoadClass, instrs: Vec<Instr>) -> Program {
|
||||
let seed_string = format!("igneum-scratch-edge/{name}");
|
||||
let seed_bytes = seed_string.as_bytes().to_vec();
|
||||
let k = instrs.iter().filter(|i| i.op == Op::Scratch).count();
|
||||
assert_eq!(k, c.scratch_slots(), "{name}: the class carries the program's scratch count");
|
||||
Program {
|
||||
seed: seed_words_from_bytes(&seed_bytes),
|
||||
seed_string,
|
||||
seed_bytes,
|
||||
generator: GENERATOR_VERSION,
|
||||
attempt: 0,
|
||||
class: c,
|
||||
instrs,
|
||||
}
|
||||
}
|
||||
|
||||
/// The edge set for a scratch of `kb` KiB per warp. Each entry: (name, what it drives, program).
|
||||
fn edge_programs(kb: u8) -> Vec<(String, &'static str, Program)> {
|
||||
let m = LoadClass::scratch(1, kb).scratch_slot_mask();
|
||||
let dsts = [2u8, 3, 4, 5, 6, 7, 0, 2, 3, 4, 5, 6, 7, 0, 2, 3];
|
||||
let scr = |n: usize, src: u8| -> Vec<Instr> { (0..n).map(|i| ins(Op::Scratch, dsts[i], src)).collect() };
|
||||
let mut v = Vec::new();
|
||||
// every RMW of the hash to slot 0 through a zero register: 64 dependent RMWs on one slot per lane
|
||||
let mut p = vec![ins(Op::Sub, 1, 1)];
|
||||
p.extend(scr(8, 1));
|
||||
v.push(("slot0".to_string(), "r1 = 0: every RMW to slot 0", edge(&format!("slot0/k{kb}"), LoadClass::scratch(8, kb), p)));
|
||||
// slot MASK through the in-range register MASK
|
||||
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m)];
|
||||
p.extend(scr(8, 1));
|
||||
v.push(("slotmask".to_string(), "r1 = MASK: every RMW to the last slot", edge(&format!("slotmask/k{kb}"), LoadClass::scratch(8, kb), p)));
|
||||
// slot MASK through the out-of-range register 0xffffffff
|
||||
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, 1), ins(Op::Sub, 1, 2)];
|
||||
p.extend(scr(8, 1));
|
||||
v.push(("ones".to_string(), "r1 = 0xffffffff: masked to the last slot", edge(&format!("ones/k{kb}"), LoadClass::scratch(8, kb), p)));
|
||||
// slot 0 through the out-of-range register MASK + 1
|
||||
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m.wrapping_add(1))];
|
||||
p.extend(scr(8, 1));
|
||||
v.push(("maskplus1".to_string(), "r1 = MASK + 1: masked to slot 0", edge(&format!("maskplus1/k{kb}"), LoadClass::scratch(8, kb), p)));
|
||||
// 16 RMWs per iteration on slot 0: 128 dependent RMWs on one slot per lane per hash
|
||||
let mut p = vec![ins(Op::Sub, 1, 1)];
|
||||
p.extend(scr(16, 1));
|
||||
v.push(("sixteen".to_string(), "16 RMWs per iteration on slot 0", edge(&format!("sixteen/k{kb}"), LoadClass::scratch(16, kb), p)));
|
||||
// one slot per lane from the init words: lanes with equal slots would show any cross-lane aliasing
|
||||
// (r5 is the slot register and is never a destination here)
|
||||
let p: Vec<Instr> = [0u8, 1, 2, 3, 4, 6, 7, 0].iter().map(|&d| ins(Op::Scratch, d, 5)).collect();
|
||||
v.push(("lanevar".to_string(), "r5 never written: one init-dependent slot per lane", edge(&format!("lanevar/k{kb}"), LoadClass::scratch(8, kb), p)));
|
||||
// alternating slot 0 and slot MASK inside one iteration
|
||||
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, m)];
|
||||
for (i, &d) in [3u8, 4, 5, 6, 7, 0, 3, 4].iter().enumerate() {
|
||||
// r1 and r2 hold the two slots and are never destinations
|
||||
p.push(ins(Op::Scratch, d, if i % 2 == 0 { 1 } else { 2 }));
|
||||
}
|
||||
v.push(("twoslots".to_string(), "slot 0 and slot MASK alternating", edge(&format!("twoslots/k{kb}"), LoadClass::scratch(8, kb), p)));
|
||||
v
|
||||
}
|
||||
|
||||
/// The hand model: a second, minimal interpreter for the ops the edge programs use (sub, add, scratch), with its
|
||||
/// own slot store keyed by (lane, slot). `mutate` swaps the rewrite's words to show the comparison has teeth.
|
||||
fn hand_model(p: &Program, base: u32, mutate: bool) -> [u64; 32] {
|
||||
let seed = &p.seed;
|
||||
let m = p.class.scratch_slot_mask();
|
||||
let mut r = [[0u32; LANES]; 8];
|
||||
for lane in 0..LANES {
|
||||
let nonce = base.wrapping_add(lane as u32);
|
||||
for i in 0..8 {
|
||||
let mut x = nonce ^ seed[i];
|
||||
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
|
||||
x = splitmix32(x);
|
||||
r[i][lane] = x ^ seed[(i + 1) & 7];
|
||||
}
|
||||
}
|
||||
let mut store: HashMap<(usize, u32), [u32; 3]> = HashMap::new();
|
||||
for _ in 0..ITERATIONS {
|
||||
let sel = r[0];
|
||||
for ins in &p.instrs {
|
||||
let (d, a) = (ins.dst as usize, ins.src as usize);
|
||||
match ins.op {
|
||||
Op::Sub => {
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] = r[d][lane].wrapping_sub(r[a][lane]);
|
||||
}
|
||||
}
|
||||
Op::Add => {
|
||||
for lane in 0..LANES {
|
||||
let c = if (sel[lane] >> ins.bit) & 1 != 0 { ins.imm2 } else { ins.imm };
|
||||
r[d][lane] = r[d][lane].wrapping_add(r[a][lane]).wrapping_add(c);
|
||||
}
|
||||
}
|
||||
Op::Scratch => {
|
||||
for lane in 0..LANES {
|
||||
let slot = r[a][lane] & m;
|
||||
let w = *store.entry((lane, slot)).or_insert_with(|| {
|
||||
let mut f = [0u32; 3];
|
||||
for j in 0..3u32 {
|
||||
// the fill, written out in full rather than through verify::scratch_fill
|
||||
let n = base.wrapping_add(lane as u32);
|
||||
f[j as usize] = splitmix32(
|
||||
(n ^ seed[j as usize])
|
||||
.wrapping_add(slot.wrapping_mul(0x9E37_79B1))
|
||||
.wrapping_add((j + 1).wrapping_mul(0x85EB_CA77)),
|
||||
);
|
||||
}
|
||||
f
|
||||
});
|
||||
let mut x = r[d][lane] ^ w[0];
|
||||
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[1];
|
||||
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[2];
|
||||
r[d][lane] = x;
|
||||
let out = if mutate {
|
||||
[x.rotate_left(7) ^ w[2], x ^ w[1], x.wrapping_add(w[0])]
|
||||
} else {
|
||||
[x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])]
|
||||
};
|
||||
store.insert((lane, slot), out);
|
||||
}
|
||||
}
|
||||
other => panic!("the hand model does not implement {other:?}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
let mut out = [0u64; 32];
|
||||
for lane in 0..LANES {
|
||||
let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21);
|
||||
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
|
||||
out[lane] = ((hi as u64) << 32) | lo as u64;
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// The four unit bases of every edge vector: 0 and 32 (two consecutive units, the pair a one-warp persistent
|
||||
/// launch runs on one arena), a unit straddling 2^31, and the unit that wraps past 2^32.
|
||||
const EDGE_BASES: [u32; 4] = [0, 32, 0x7fff_fff0, 0xffff_ffe0];
|
||||
|
||||
#[test]
|
||||
fn edge_programs_match_the_hand_model() {
|
||||
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24);
|
||||
let mut cases = 0;
|
||||
for kb in [32u8, 128] {
|
||||
for (name, what, p) in edge_programs(kb) {
|
||||
let slots = p.class.scratch_slots_per_lane();
|
||||
for base in EDGE_BASES {
|
||||
let (res, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true);
|
||||
let hand = hand_model(&p, base, false);
|
||||
assert_eq!(res.hashes, hand, "{name} k{kb} base {base:#x}: interpreter against the hand model ({what})");
|
||||
assert_ne!(res.hashes, hand_model(&p, base, true), "{name} k{kb}: the comparison has teeth");
|
||||
// the slots the trace saw are the ones the program was built to drive
|
||||
let slot_set: std::collections::BTreeSet<u32> = ev.iter().map(|e| e.slot).collect();
|
||||
let m = (slots - 1) as u32;
|
||||
match name.as_str() {
|
||||
"slot0" | "maskplus1" | "sixteen" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![0]),
|
||||
"slotmask" | "ones" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![m]),
|
||||
"twoslots" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![0, m]),
|
||||
"lanevar" => {
|
||||
for e in &ev {
|
||||
assert!(e.slot <= m);
|
||||
}
|
||||
}
|
||||
_ => unreachable!(),
|
||||
}
|
||||
// the chain depth on the driven slot: every RMW after the first per lane is a re-hit
|
||||
let per_lane = p.scratch_ops_per_hash();
|
||||
let hits = ev.iter().filter(|e| e.hit).count();
|
||||
let expected_hits = match name.as_str() {
|
||||
"twoslots" => (per_lane - 2) * LANES,
|
||||
_ => (per_lane - 1) * LANES,
|
||||
};
|
||||
assert_eq!(hits, expected_hits, "{name} k{kb}: re-hits");
|
||||
cases += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
assert_eq!(cases, 2 * 7 * 4);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
// 4. The static scratch check over every emitted kernel of every scr pack (question 4)
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum Dialect {
|
||||
Metal,
|
||||
Cuda,
|
||||
OpenCl,
|
||||
}
|
||||
|
||||
/// The static scratch check: every scratch read-modify-write in an emitted kernel has the one masked form the
|
||||
/// emitter writes, the arena is the lane's own `slots x 4` words, the tag is `salt + unit`, and nothing else
|
||||
/// touches the scratch. Like the dataset mask check of `TESTS.md` section 5 and `tests/packs.rs`, a text check:
|
||||
/// the guarantee is that the emitter has one template and it masks.
|
||||
pub fn scratch_text_check(text: &str, dialect: Dialect, k: usize, slots: usize, kernels: usize) -> Result<(), String> {
|
||||
assert!(kernels >= 1);
|
||||
// every count below is per hash kernel; an OpenCL bound file carries igneum_hash and igneum_hash_bound
|
||||
let k = k * kernels;
|
||||
assert!(slots.is_power_of_two() && slots >= 1);
|
||||
let mask = (slots - 1) as u32;
|
||||
let wpl = slots * 4;
|
||||
let (u, load, store, ptr) = match dialect {
|
||||
Dialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "device uint* arena"),
|
||||
Dialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "uint32_t* arena"),
|
||||
Dialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); }", "__global uint* arena"),
|
||||
};
|
||||
let count = |needle: &str| text.matches(needle).count();
|
||||
let mut errs = Vec::new();
|
||||
let mut expect = |what: &str, got: usize, want: usize| {
|
||||
if got != want {
|
||||
errs.push(format!("{what}: {got}, expected {want}"));
|
||||
}
|
||||
};
|
||||
// k slot computations, each masked with exactly the class's mask and immediately followed by the one load form
|
||||
expect("slot definitions `{ u s_ = r`", count(&format!("{{ {u} s_ = r")), k);
|
||||
expect("masked slot followed by the load", count(&format!(" & {mask}u; {load}")), k);
|
||||
expect("stores of the tagged slot", count(store), k);
|
||||
expect("tag compares", count("(v_.x == tag)"), k);
|
||||
expect("fill calls (three per RMW)", count("scr_fill(gbase, lane, s_, "), 3 * k);
|
||||
// the arena: one definition with the class's words per lane, and 2k uses (one load, one store per RMW)
|
||||
expect("arena definition", count(&format!("{ptr} = scratch + ((size_t)warp_ * 32u + lane) * {wpl}u;")), kernels);
|
||||
expect("arena mentions (definition + load + store per RMW)", count("arena"), kernels + 2 * k);
|
||||
expect("tag definition `tag = salt + g_`", count(&format!("{u} tag = salt + g_;")), kernels);
|
||||
expect("direct scratch indexing", count("scratch["), 0);
|
||||
expect("scratch pointer arithmetic outside the arena definition", count("scratch +"), kernels);
|
||||
// no other mask value on a slot: every `s_ = r` line carries the class mask and nothing else carries ` & Nu; uint4 v_`
|
||||
let any_mask_load = count(&format!("u; {load}"));
|
||||
expect("loads preceded by some mask (must all be the class mask)", any_mask_load, k);
|
||||
if errs.is_empty() {
|
||||
Ok(())
|
||||
} else {
|
||||
Err(errs.join("; "))
|
||||
}
|
||||
}
|
||||
|
||||
fn packs_rw_dir() -> PathBuf {
|
||||
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-readwidth")
|
||||
}
|
||||
|
||||
fn scr_packs() -> Vec<String> {
|
||||
let mut v: Vec<String> = std::fs::read_dir(packs_rw_dir())
|
||||
.unwrap()
|
||||
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
|
||||
.filter(|n| n.starts_with("scr"))
|
||||
.collect();
|
||||
v.sort();
|
||||
v
|
||||
}
|
||||
|
||||
fn read_pack(pack: &str, file: &str) -> String {
|
||||
let p = packs_rw_dir().join(pack).join(file);
|
||||
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
|
||||
}
|
||||
|
||||
/// Every scr pack regenerates from its program.json (seed bytes, class, day bytes, size) to the same six kernel
|
||||
/// texts, byte for byte, and every one of those texts passes the static scratch check for the class's k and slot
|
||||
/// count; the check fails on four deliberate breaks of a copy of the Metal text (mask dropped, mask changed, arena
|
||||
/// stride changed, a stray scratch access) and on the OpenCL and CUDA twins of the first.
|
||||
#[test]
|
||||
fn scr_packs_regenerate_and_pass_the_static_scratch_check() {
|
||||
let packs = scr_packs();
|
||||
assert!(packs.len() >= 6, "the scr packs: {packs:?}");
|
||||
let mut checked = 0;
|
||||
let mut sample_metal = String::new();
|
||||
let mut sample_cl = String::new();
|
||||
let mut sample_cu = String::new();
|
||||
let mut sample_k = 0;
|
||||
let mut sample_slots = 0;
|
||||
for pack in &packs {
|
||||
let j: Value = serde_json::from_str(&read_pack(pack, "program.json")).unwrap();
|
||||
let name = j["load_class"].as_str().unwrap();
|
||||
let c = class(name);
|
||||
assert_eq!(&format!("{name}"), pack, "pack directory named after its class");
|
||||
let seed = j["seed"].as_str().unwrap();
|
||||
let seed_bytes = igneum_pow::bind::unhex(j["seed_bytes"].as_str().unwrap()).unwrap();
|
||||
let day_bytes = igneum_pow::bind::unhex(j["dataset"]["day_bytes"].as_str().unwrap()).unwrap();
|
||||
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
|
||||
assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard");
|
||||
let program = generate_from_seed_bytes_class(seed, &seed_bytes, c);
|
||||
assert_eq!(program.class, c);
|
||||
assert_eq!(program.program_id(), u64::from_str_radix(j["program_id"].as_str().unwrap().trim_start_matches("0x"), 16).unwrap());
|
||||
let mut dataset = DatasetSource::from_key(seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2);
|
||||
dataset.key_bytes = day_bytes;
|
||||
let e = Epoch { program, dataset };
|
||||
let p = &e.program;
|
||||
let mp = e.dataset.memhard().map(|m| &m.params);
|
||||
let k = c.scratch_slots();
|
||||
let slots = c.scratch_slots_per_lane();
|
||||
assert_eq!(p.scratch_ops_per_hash(), k * ITERATIONS);
|
||||
for (file, text, dialect, kernels) in [
|
||||
("program.metal", metal_program(p, log2, LoadSource::Stored), Dialect::Metal, 1),
|
||||
("program_bound.metal", metal_program_bound(p, log2), Dialect::Metal, 1),
|
||||
("kernel.cu", cuda_kernel(p, mp), Dialect::Cuda, 1),
|
||||
("kernel_bound.cu", cuda_kernel_bound(p, mp), Dialect::Cuda, 1),
|
||||
("kernel.cl", opencl_kernel(p, mp), Dialect::OpenCl, 1),
|
||||
// the OpenCL bound file carries igneum_hash and igneum_hash_bound
|
||||
("kernel_bound.cl", opencl_kernel_bound(p, mp), Dialect::OpenCl, 2),
|
||||
] {
|
||||
let on_disk = read_pack(pack, file);
|
||||
assert_eq!(on_disk, text, "{pack}/{file}: the pack is the emitter's text");
|
||||
// scr0 is the persistent control: an arena and a tag, no read-modify-write; the check holds with k = 0
|
||||
scratch_text_check(&on_disk, dialect, k, slots, kernels).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"));
|
||||
checked += 1;
|
||||
}
|
||||
// the vectors of the pack are the CPU's
|
||||
let v: Value = serde_json::from_str(&read_pack(pack, "vectors.json")).unwrap();
|
||||
for w in v["warps"].as_array().unwrap() {
|
||||
let base = w["base_nonce"].as_u64().unwrap() as u32;
|
||||
let got = e.hash_warp(base);
|
||||
for (lane, x) in w["expected"].as_array().unwrap().iter().enumerate() {
|
||||
let want = u64::from_str_radix(x.as_str().unwrap().trim_start_matches("0x"), 16).unwrap();
|
||||
assert_eq!(got[lane], want, "{pack}: base {base} lane {lane}");
|
||||
}
|
||||
}
|
||||
if k == 4 && slots == 64 {
|
||||
sample_metal = read_pack(pack, "program.metal");
|
||||
sample_cl = read_pack(pack, "kernel.cl");
|
||||
sample_cu = read_pack(pack, "kernel.cu");
|
||||
sample_k = k;
|
||||
sample_slots = slots;
|
||||
}
|
||||
}
|
||||
assert_eq!(checked, packs.len() * 6);
|
||||
println!("static scratch check: {checked} kernels over {} scr packs", packs.len());
|
||||
|
||||
// The deliberate breaks (the watcher rule of CLAUDE.md: a check is trusted once it fails on a known-broken
|
||||
// case). Each must be caught; the message names what.
|
||||
assert!(sample_k == 4 && sample_slots == 64, "scr4k32 is in the pack set");
|
||||
let mask = format!(" & {}u; uint4 v_", sample_slots - 1);
|
||||
let broken_mask = sample_metal.replacen(&mask, "; uint4 v_", 1);
|
||||
assert_ne!(broken_mask, sample_metal);
|
||||
let e = scratch_text_check(&broken_mask, Dialect::Metal, 4, 64, 1).unwrap_err();
|
||||
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
|
||||
println!("break 1 (one mask dropped, Metal): {e}");
|
||||
let wrong_mask = sample_metal.replace(" & 63u;", " & 127u;");
|
||||
let e = scratch_text_check(&wrong_mask, Dialect::Metal, 4, 64, 1).unwrap_err();
|
||||
assert!(e.contains("masked slot followed by the load: 0, expected 4"), "{e}");
|
||||
println!("break 2 (mask 63 -> 127 on every RMW, Metal): {e}");
|
||||
let wrong_stride = sample_metal.replace("* 256u;", "* 128u;");
|
||||
let e = scratch_text_check(&wrong_stride, Dialect::Metal, 4, 64, 1).unwrap_err();
|
||||
assert!(e.contains("arena definition: 0, expected 1"), "{e}");
|
||||
println!("break 3 (arena stride 256 -> 128 words, Metal): {e}");
|
||||
let stray = format!("{sample_metal}\n// stray\n// arena[0] = 0u; scratch[1] = 1u;\n");
|
||||
let e = scratch_text_check(&stray, Dialect::Metal, 4, 64, 1).unwrap_err();
|
||||
assert!(e.contains("arena mentions") && e.contains("direct scratch indexing: 1, expected 0"), "{e}");
|
||||
println!("break 4 (a stray arena and scratch access, Metal): {e}");
|
||||
let e = scratch_text_check(&sample_cl.replacen(" & 63u; uint4 v_ = vload4", "; uint4 v_ = vload4", 1), Dialect::OpenCl, 4, 64, 1).unwrap_err();
|
||||
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
|
||||
println!("break 5 (one mask dropped, OpenCL): {e}");
|
||||
let e = scratch_text_check(&sample_cu.replacen(" & 63u; uint4 v_ = *(const uint4*)", "; uint4 v_ = *(const uint4*)", 1), Dialect::Cuda, 4, 64, 1).unwrap_err();
|
||||
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
|
||||
println!("break 6 (one mask dropped, CUDA): {e}");
|
||||
// and the unbroken texts pass under the same calls
|
||||
scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 1).unwrap();
|
||||
scratch_text_check(&sample_cl, Dialect::OpenCl, 4, 64, 1).unwrap();
|
||||
scratch_text_check(&sample_cu, Dialect::Cuda, 4, 64, 1).unwrap();
|
||||
// a wrong slot count, RMW count or kernel count against a right text fails too (the check is tied to the class)
|
||||
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 256, 1).is_err());
|
||||
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 3, 64, 1).is_err());
|
||||
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 2).is_err());
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
// 5. The fuzz: 200 generated scratch programs, contract on every instruction, 4 units each across the 32-bit
|
||||
// range including the wrap; with IGNEUM_SCRATCH_PACKS_OUT the packs for the Metal runs (question 3, 4)
|
||||
// ---------------------------------------------------------------------------------------------------------------
|
||||
|
||||
/// Write a pack whose vectors.json carries `bases` (any number of units) instead of the three standard bases.
|
||||
fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) -> Vec<[u64; 32]> {
|
||||
let mut pack = export_pack(e, day, source);
|
||||
let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect();
|
||||
let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true);
|
||||
for f in pack.files.iter_mut() {
|
||||
if f.0 == "vectors.json" {
|
||||
f.1 = vj.clone();
|
||||
}
|
||||
}
|
||||
pack.write_to(dir).unwrap();
|
||||
outs
|
||||
}
|
||||
|
||||
fn contract(p: &Program) {
|
||||
assert_eq!(p.instrs.len(), INSTR_COUNT);
|
||||
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count() + p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), 16);
|
||||
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), p.class.scratch_slots());
|
||||
assert!(p.instrs[0].op != Op::Load && p.instrs[0].op != Op::Scratch, "instruction 0 is never a memory op");
|
||||
for (k, i) in p.instrs.iter().enumerate() {
|
||||
assert!(i.src != i.dst, "#{k}: src == dst");
|
||||
assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot);
|
||||
assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask);
|
||||
assert!(i.dst < 8 && i.src < 8 && i.src2 < 8);
|
||||
assert_eq!(i.width, 1, "#{k}: a scratch class reads one-word loads");
|
||||
}
|
||||
assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fuzz_scr_programs_cpu() {
|
||||
let n: usize = std::env::var("IGNEUM_SCRATCH_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200);
|
||||
let out = std::env::var("IGNEUM_SCRATCH_PACKS_OUT").ok().map(PathBuf::from);
|
||||
let mut rng = SplitMix64::new(0x6967_6e65_756d_2d73); // "igneum-s"
|
||||
let day = "2026-10-03";
|
||||
let closed = DatasetSource::new(day, DatasetMode::ClosedForm, 28);
|
||||
// memory-hard sources per size, built once each (the cache fill is 0.2 s); only when packs are written
|
||||
let mut mh: HashMap<u32, DatasetSource> = HashMap::new();
|
||||
let mut manifest = String::from("pack\tclass\tlog2\tprogram_id\tscratch_ops_per_hash\tbases\n");
|
||||
let mut per_class: HashMap<String, usize> = HashMap::new();
|
||||
let mut units = 0usize;
|
||||
let mut wraps = 0usize;
|
||||
if let Some(dir) = &out {
|
||||
std::fs::create_dir_all(dir).unwrap();
|
||||
// the edge packs first: 64 MiB datasets (no dataset load in them), the four edge bases
|
||||
for kb in [32u8, 128] {
|
||||
for (name, _what, p) in edge_programs(kb) {
|
||||
let log2 = 24;
|
||||
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2));
|
||||
let e = Epoch { program: p, dataset: ds };
|
||||
let pack_name = format!("edge-{name}-k{kb}");
|
||||
write_pack_with_bases(&dir.join(&pack_name), &e, day, &EDGE_BASES, "igneum-pow tests/scratch.rs edge");
|
||||
manifest.push_str(&format!(
|
||||
"{pack_name}\t{}\t{log2}\t{:016x}\t{}\t{}\n",
|
||||
e.program.class.name(),
|
||||
e.program.program_id(),
|
||||
e.program.scratch_ops_per_hash(),
|
||||
EDGE_BASES.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
|
||||
));
|
||||
mh.insert(log2, e.dataset);
|
||||
}
|
||||
}
|
||||
}
|
||||
for i in 0..n {
|
||||
let name = CLASSES[rng.below(CLASSES.len() as u64) as usize];
|
||||
let c = class(name);
|
||||
let seed = format!("igneum-scratch-fuzz/{i}");
|
||||
let p = generate_class(&seed, c);
|
||||
contract(&p);
|
||||
*per_class.entry(name.to_string()).or_insert(0) += 1;
|
||||
// four bases: one inside a 256-nonce batch (in-batch check on the GPU), one straddling 2^31, one in
|
||||
// the last 256 nonces (the unit wraps past 2^32 or ends on it), one uniform
|
||||
let b0 = (rng.below(8) as u32) * 32;
|
||||
let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32);
|
||||
let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32);
|
||||
let b3 = (rng.next() as u32) & !31;
|
||||
let bases = [b0, b1, b2, b3];
|
||||
// an aligned unit never straddles 2^32 (spec 1.9); the top unit ends on 0xffffffff and the persistent
|
||||
// kernel's unit sequence wraps inside a launch, which the Metal run checks with packbench --batch-base
|
||||
wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count();
|
||||
// the CPU: the interpreter is deterministic and every scratch event is inside the lane's slots
|
||||
for &b in &bases {
|
||||
let (r1, ev) = interpret_warp_scratch(&p, &p.seed, b, &closed, true);
|
||||
let r2 = interpret_warp_scratch(&p, &p.seed, b, &closed, false).0;
|
||||
assert_eq!(r1.hashes, r2.hashes);
|
||||
assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES);
|
||||
assert!(ev.iter().all(|e: &ScratchEvent| e.slot < c.scratch_slots_per_lane() as u32));
|
||||
units += 1;
|
||||
}
|
||||
if let Some(dir) = &out {
|
||||
let log2 = [24u32, 26, 28][rng.below(3) as usize];
|
||||
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2));
|
||||
let e = Epoch { program: p, dataset: ds };
|
||||
let pack_name = format!("fuzz-{i:03}-{name}-l{log2}");
|
||||
write_pack_with_bases(&dir.join(&pack_name), &e, day, &bases, "igneum-pow tests/scratch.rs fuzz");
|
||||
manifest.push_str(&format!(
|
||||
"{pack_name}\t{name}\t{log2}\t{:016x}\t{}\t{}\n",
|
||||
e.program.program_id(),
|
||||
e.program.scratch_ops_per_hash(),
|
||||
bases.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
|
||||
));
|
||||
mh.insert(log2, e.dataset);
|
||||
} else {
|
||||
let _ = rng.below(3);
|
||||
}
|
||||
}
|
||||
let mut classes: Vec<_> = per_class.iter().collect();
|
||||
classes.sort();
|
||||
println!("fuzz: {n} programs, {units} units on the CPU, {wraps} units in the top 256 nonces, classes {classes:?}");
|
||||
assert_eq!(units, 4 * n);
|
||||
assert_eq!(wraps, n, "every program has a unit in the top 256 nonces");
|
||||
if let Some(dir) = &out {
|
||||
std::fs::write(dir.join("manifest.tsv"), manifest).unwrap();
|
||||
println!("packs written to {}", dir.display());
|
||||
}
|
||||
}
|
||||
|
||||
/// The fold and rewrite, restated: a slot after `d` dependent RMWs holds 96 bits that are a function of the fill
|
||||
/// (3 words, a pure function of nonce, slot and seed) and the `d` fold values; a chip that keeps the `d` fold
|
||||
/// values (32 bits each) instead of the 96-bit slot recomputes the slot in `d` rewrites. This test pins the
|
||||
/// arithmetic the analysis uses (question 2): the replay from the fold values reproduces the slot.
|
||||
#[test]
|
||||
fn slot_is_replayable_from_its_fold_values() {
|
||||
let seed = seed_words_from_bytes(b"igneum-genesis");
|
||||
let (base, lane, slot) = (0x1234_5600u32, 5u32, 17u32);
|
||||
let fill = [scratch_fill(&seed, base, lane, slot, 0), scratch_fill(&seed, base, lane, slot, 1), scratch_fill(&seed, base, lane, slot, 2)];
|
||||
let mut rng = SplitMix64::new(99);
|
||||
let dsts: Vec<u32> = (0..64).map(|_| rng.next() as u32).collect();
|
||||
// the honest sequence: read, fold, rewrite, 64 times
|
||||
let mut w = fill;
|
||||
let mut xs = Vec::new();
|
||||
for &d in &dsts {
|
||||
let x = fold_words(d, &w);
|
||||
xs.push(x);
|
||||
w = scratch_rewrite(x, &w);
|
||||
}
|
||||
// the replay: from the fill and the stored fold values alone
|
||||
let mut w2 = fill;
|
||||
for &x in &xs {
|
||||
w2 = scratch_rewrite(x, &w2);
|
||||
}
|
||||
assert_eq!(w, w2);
|
||||
// and nothing shorter: the fold value at step d depends on the slot content at step d, which depends on
|
||||
// every earlier fold value (drop one and the chain diverges)
|
||||
let mut w3 = fill;
|
||||
for (i, &x) in xs.iter().enumerate() {
|
||||
if i != 10 {
|
||||
w3 = scratch_rewrite(x, &w3);
|
||||
}
|
||||
}
|
||||
assert_ne!(w, w3);
|
||||
}
|
||||
|
|
@ -5,7 +5,7 @@
|
|||
// against the Rust CPU reference and timed without a Swift mirror of the generator. One file, no packages.
|
||||
//
|
||||
// swiftc -O -target arm64-apple-macos11 -o packbench packbench.swift -framework Metal
|
||||
// ./packbench --pack <dir> [--batches 5] [--batch-log2 24] [--group 256] [--warps 2048]
|
||||
// ./packbench --pack <dir> [--batches 5] [--batch-log2 24] [--group 256] [--warps 2048] [--batch-base 0]
|
||||
//
|
||||
// Prints one RESULT line per run: vectors, cache and dataset checks, the batch fingerprint (FNV-1a 64 over the 2^B
|
||||
// outputs at base nonce 0) and MH/s by wall and by GPU time. Variant 5 packs (IGNEUM_PERSISTENT_WARPS) are launched
|
||||
|
|
@ -16,7 +16,7 @@ import Metal
|
|||
func nowMs() -> Double { return Double(DispatchTime.now().uptimeNanoseconds) / 1e6 }
|
||||
func fail(_ m: String) -> Never { print("FAIL: \(m)"); exit(1) }
|
||||
|
||||
struct Opts { var pack = ""; var batches = 5; var batchLog2 = 24; var group = 256; var warps = 2048 }
|
||||
struct Opts { var pack = ""; var batches = 5; var batchLog2 = 24; var group = 256; var warps = 2048; var batchBase: UInt32 = 0 }
|
||||
var opts = Opts()
|
||||
var args = Array(CommandLine.arguments.dropFirst())
|
||||
while !args.isEmpty {
|
||||
|
|
@ -28,6 +28,7 @@ while !args.isEmpty {
|
|||
case "--batch-log2": opts.batchLog2 = Int(next())!
|
||||
case "--group": opts.group = Int(next())!
|
||||
case "--warps": opts.warps = Int(next())!
|
||||
case "--batch-base": opts.batchBase = UInt32(next())! // base nonce of the fingerprint batch (default 0; a base near 2^32 makes the persistent unit sequence wrap inside the launch)
|
||||
default: fail("unknown argument \(a)")
|
||||
}
|
||||
}
|
||||
|
|
@ -183,12 +184,13 @@ for (i, base) in vecBases.enumerated() {
|
|||
var nonces = 1 << opts.batchLog2
|
||||
if persistent { let unit = 32 * min(warpsN, nonces / 32); nonces = (nonces / unit) * unit }
|
||||
let out = device.makeBuffer(length: nonces * 8, options: .storageModeShared)!
|
||||
let (warmWall, warmGpu) = run { enc in encodeHash(enc, out: out, base: 0, nonces: nonces, group: opts.group) }
|
||||
let (warmWall, warmGpu) = run { enc in encodeHash(enc, out: out, base: opts.batchBase, nonces: nonces, group: opts.group) }
|
||||
let outPtr = out.contents().bindMemory(to: UInt64.self, capacity: nonces)
|
||||
var batchVecPass = 0, batchVecN = 0
|
||||
for (i, base) in vecBases.enumerated() where Int(base) + 32 <= nonces {
|
||||
for (i, base) in vecBases.enumerated() where Int(base &- opts.batchBase) + 32 <= nonces { // the batch window, wrapping past 2^32
|
||||
let off = Int(base &- opts.batchBase)
|
||||
batchVecN += 1
|
||||
if (0..<32).allSatisfy({ outPtr[Int(base) + $0] == vecOuts[i][$0] }) { batchVecPass += 1 }
|
||||
if (0..<32).allSatisfy({ outPtr[off + $0] == vecOuts[i][$0] }) { batchVecPass += 1 }
|
||||
}
|
||||
let fingerprint = fnv1a64(out.contents(), nonces * 8)
|
||||
// Timed batches
|
||||
|
|
@ -205,5 +207,5 @@ print("device \(device.name); compile \(String(format: "%.0f", compileMs)) ms; c
|
|||
print("cache FNV-1a 64 \(String(format: "%016llx", cacheFnv)) \(cacheOk ? "PASS" : "FAIL"); dataset head and last \(dsOk ? "PASS" : "FAIL"); vectors standalone \(vecPass)/\(vecBases.count), in batch \(batchVecPass)/\(batchVecN)")
|
||||
print("warm-up batch \(nonces) hashes: \(String(format: "%.1f", warmGpu)) ms GPU, \(String(format: "%.1f", warmWall)) ms wall")
|
||||
let overall = cacheOk && dsOk && vecPass == vecBases.count && batchVecPass == batchVecN
|
||||
print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) arena_mib=\(persistent ? warpsN : 0) nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")")
|
||||
print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) batch_base=\(opts.batchBase) arena_mib=\(persistent ? warpsN : 0) nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")")
|
||||
exit(overall ? 0 : 1)
|
||||
|
|
|
|||
Loading…
Reference in a new issue