diff --git a/igneum-pow/src/accept.rs b/igneum-pow/src/accept.rs index 7f8c4438a..aca1702ea 100644 --- a/igneum-pow/src/accept.rs +++ b/igneum-pow/src/accept.rs @@ -14,7 +14,7 @@ //! costs about a millisecond on one core. The census (section 7.3) checked on 100,000 programs that the //! closed-form verdict agrees with the memory-hard one on all but 39 threshold-edge cases. -use crate::generator::{Instr, Op, Program, INSTR_COUNT, ITERATIONS, LANES, SCRATCH_SLOT_MASK}; +use crate::generator::{Instr, Op, Program, INSTR_COUNT, ITERATIONS, LANES}; use crate::seed::{fnv1a64, SplitMix64}; use crate::verify::{dataset_elem, fold_words, splitmix32, ScratchModel}; @@ -33,8 +33,10 @@ pub const BIAS_TOLERANCE: u32 = 136; /// Distinct addresses per lane per evaluation, summed over 2,048 evaluations, must exceed this (mean above 120). pub const MIN_DISTINCT_SUM: u64 = 245_760; -/// The distinct-address bound for a program with `loads` loads per hash: the same 120 of 128 ratio, so +/// The distinct-address bound for a program with `loads` dataset loads per hash: the same 120 of 128 ratio, so /// [`MIN_DISTINCT_SUM`] for the lottery hash and `loads x 1,920` for the read-width classes with other counts. +/// Variant 5's scratch read-modify-writes are not dataset loads: their slots repeat by design (a later +/// read-modify-write sees an earlier write), so they are neither counted nor bounded here. pub fn min_distinct_sum(loads: usize) -> u64 { loads as u64 * ACCEPT_HASHES as u64 * 120 / 128 } @@ -73,7 +75,7 @@ impl std::fmt::Display for Reject { Reject::Saturated { count } => write!(f, "(c) {count} of 16384 final register values saturated (limit 163)"), Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"), Reject::DistinctAddresses { sum } => { - write!(f, "(c) distinct addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 loads)", *sum as f64 / 2048.0) + write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0) } } } @@ -189,7 +191,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut } let mut idx = [0u32; LANES]; let mut nload = 0usize; - let mut scratch = if p.has_scratch() { Some(ScratchModel::new()) } else { None }; + let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None }; + let slot_mask = p.class.scratch_slot_mask(); for it in 0..ITERATIONS { let sel = r[0]; for (k, ins) in p.instrs.iter().enumerate() { @@ -200,7 +203,7 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut // Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word). let m = scratch.as_mut().expect("a scratch op needs a scratch class"); for lane in 0..LANES { - idx[lane] = r[a][lane] & SCRATCH_SLOT_MASK; + idx[lane] = r[a][lane] & slot_mask; } if idx.iter().all(|&x| x == idx[0]) { return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 }); @@ -332,7 +335,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut sl.sort_unstable(); let mut distinct = 0u64; for k in 0..loads { - if k == 0 || sl[k] != sl[k - 1] { + // scratch slots carry bit 31 (variant 5) and are not dataset addresses + if sl[k] & 0x8000_0000 == 0 && (k == 0 || sl[k] != sl[k - 1]) { distinct += 1; } } @@ -367,7 +371,7 @@ pub fn check_dynamic(p: &Program) -> Result { } bias_max = bias_max.max(d); } - if acc.distinct_sum <= min_distinct_sum(loads) { + if acc.distinct_sum <= min_distinct_sum(loads - p.scratch_ops_per_hash()) { return Err(Reject::DistinctAddresses { sum: acc.distinct_sum }); } Ok(AcceptReport { distinct_sum: acc.distinct_sum, saturated: acc.saturated, bias_max }) @@ -395,7 +399,7 @@ mod tests { /// with `verify.rs` on every class (the fold is shared, the addresses are aligned the same way). #[test] fn classes_pass_and_match_verify() { - for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2", "scr8"] { + for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2k32", "scr8k128"] { let c = LoadClass::parse(name).unwrap(); let p = generate_class("igneum-genesis", c); assert!(check(&p).is_ok(), "{name}"); diff --git a/igneum-pow/src/emit.rs b/igneum-pow/src/emit.rs index 4ea7a9f03..77029f280 100644 --- a/igneum-pow/src/emit.rs +++ b/igneum-pow/src/emit.rs @@ -15,7 +15,6 @@ use crate::memhard::{ CACHE_TAG, CACHE_WORDS, CHACHA_ROUNDS, CHACHA_SIGMA, ITEM_ROUNDS, }; use crate::seed::SplitMix64; -use crate::generator::{SCRATCH_BYTES_PER_WARP, SCRATCH_SLOTS, SCRATCH_SLOT_MASK, SCRATCH_WORDS_PER_LANE}; use crate::verify::{DatasetMode, DatasetSource, Epoch, FOLD_MUL, FOLD_ROT}; /// Where the words of a wide load come from (read-width experiment). @@ -127,15 +126,11 @@ fn scratch_prelude(p: &Program, dialect: CoreDialect) -> String { CoreDialect::OpenCl => ("uint", "static inline"), }; let mut s = String::new(); - s.push_str(&format!("// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, {} slots of -", SCRATCH_SLOTS)); - s.push_str("// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not -"); - s.push_str("// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. -"); + s.push_str(&format!("// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a {} KiB scratch per warp, {} slots of\n", p.class.scratch_kb, p.class.scratch_slots_per_lane())); + s.push_str("// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not\n"); + s.push_str("// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.\n"); s.push_str(&format!( - "{fn_} {u} scr_fill({u} gbase, {u} lane, {u} slot, {u} j) {{ {u} sw = (j == 0u) ? {} : ((j == 1u) ? {} : {}); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }} -", + "{fn_} {u} scr_fill({u} gbase, {u} lane, {u} slot, {u} j) {{ {u} sw = (j == 0u) ? {} : ((j == 1u) ? {} : {}); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }}\n", hex(p.seed[0]), hex(p.seed[1]), hex(p.seed[2]) @@ -145,15 +140,14 @@ fn scratch_prelude(p: &Program, dialect: CoreDialect) -> String { /// Variant 5: one scratch read-modify-write as a statement block. `arena`, `tag`, `gbase` and `lane` are in scope /// (the persistent prologue). Reads 16 bytes, folds the three data words into dst, rewrites the slot behind the tag. -fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str) -> String { +fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str, slot_mask: u32) -> String { let (u, load, store) = match dialect { CoreDialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"), CoreDialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"), CoreDialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena);"), }; format!( - "{{ {u} s_ = {a} & {}u; {load} {u} m_ = (v_.x == tag) ? 0xffffffffu : 0u; {u} w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); {u} w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); {u} w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); {u} x_ = {d} ^ w0_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w1_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w2_; {d} = x_; {store} }}", - SCRATCH_SLOT_MASK, + "{{ {u} s_ = {a} & {slot_mask}u; {load} {u} m_ = (v_.x == tag) ? 0xffffffffu : 0u; {u} w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); {u} w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); {u} w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); {u} x_ = {d} ^ w0_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w1_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w2_; {d} = x_; {store} }}", k = hex(FOLD_MUL) ) } @@ -162,29 +156,21 @@ fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str) -> String { /// choice); warp `w` owns arena `w` and runs the units `w, w + N, w + 2N, ...` of the launch. Inside the loop the /// lottery hash's text is unchanged: `gid` is the unit's first output index plus the lane. The host MUST launch /// `groups` as a multiple of N (a uniform trip count: the OpenCL local-memory exchange carries a barrier). -fn persistent_prologue(dialect: CoreDialect) -> String { +fn persistent_prologue(dialect: CoreDialect, words_per_lane: usize) -> String { let (u, tid, nthreads, ptr) = match dialect { CoreDialect::Metal => ("uint", "tid", "nthreads", "device uint*"), CoreDialect::Cuda => ("uint32_t", "(blockIdx.x * blockDim.x + threadIdx.x)", "(gridDim.x * blockDim.x)", "uint32_t*"), CoreDialect::OpenCl => ("uint", "(uint)get_global_id(0)", "(uint)get_global_size(0)", "__global uint*"), }; let mut s = String::new(); - s.push_str(&format!(" {u} lane = {tid} & 31u; -")); - s.push_str(&format!(" {u} warp_ = {tid} >> 5; -")); - s.push_str(&format!(" {u} nwarps_ = {nthreads} >> 5; -")); - s.push_str(&format!(" {ptr} arena = scratch + ((size_t)warp_ * 32u + lane) * {}u; -", SCRATCH_WORDS_PER_LANE)); - s.push_str(&format!(" for ({u} g_ = warp_; g_ < groups; g_ += nwarps_) {{ -")); - s.push_str(&format!(" {u} gid = g_ * 32u + lane; -")); - s.push_str(&format!(" {u} gbase = baseNonce + g_ * 32u; -")); - s.push_str(&format!(" {u} tag = salt + g_; -")); + s.push_str(&format!(" {u} lane = {tid} & 31u;\n")); + s.push_str(&format!(" {u} warp_ = {tid} >> 5;\n")); + s.push_str(&format!(" {u} nwarps_ = {nthreads} >> 5;\n")); + s.push_str(&format!(" {ptr} arena = scratch + ((size_t)warp_ * 32u + lane) * {words_per_lane}u;\n")); + s.push_str(&format!(" for ({u} g_ = warp_; g_ < groups; g_ += nwarps_) {{\n")); + s.push_str(&format!(" {u} gid = g_ * 32u + lane;\n")); + s.push_str(&format!(" {u} gbase = baseNonce + g_ * 32u;\n")); + s.push_str(&format!(" {u} tag = salt + g_;\n")); s } @@ -194,20 +180,13 @@ fn scratch_header_lines(p: &Program) -> String { return String::new(); } let mut s = String::new(); - s.push_str("// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch, -"); - s.push_str("// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit). -"); - s.push_str("#define IGNEUM_PERSISTENT_WARPS 1 -"); - s.push_str(&format!("#define IGNEUM_SCRATCH_OPS {} // scratch read-modify-writes per program ({} per hash) -", p.class.scratch_slots(), p.scratch_ops_per_hash())); - s.push_str(&format!("#define IGNEUM_SCRATCH_SLOTS {SCRATCH_SLOTS}u -")); - s.push_str(&format!("#define IGNEUM_SCRATCH_WORDS_PER_LANE {SCRATCH_WORDS_PER_LANE}u -")); - s.push_str(&format!("#define IGNEUM_SCRATCH_BYTES_PER_WARP {SCRATCH_BYTES_PER_WARP}u -")); + s.push_str(&format!("// Variant 5: persistent warps, a {} KiB scratch per launched warp (the host launches N warps and passes scratch,\n", p.class.scratch_kb)); + s.push_str("// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).\n"); + s.push_str("#define IGNEUM_PERSISTENT_WARPS 1\n"); + s.push_str(&format!("#define IGNEUM_SCRATCH_OPS {} // scratch read-modify-writes per program ({} per hash)\n", p.class.scratch_slots(), p.scratch_ops_per_hash())); + s.push_str(&format!("#define IGNEUM_SCRATCH_SLOTS {}u\n", p.class.scratch_slots_per_lane())); + s.push_str(&format!("#define IGNEUM_SCRATCH_WORDS_PER_LANE {}u\n", p.class.scratch_words_per_lane())); + s.push_str(&format!("#define IGNEUM_SCRATCH_BYTES_PER_WARP {}u\n", p.class.scratch_bytes_per_warp())); s } @@ -472,7 +451,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: s.push_str(&format!(" constant uint& salt [[buffer({})]],\n", b + 2)); s.push_str(" uint tid [[thread_position_in_grid]],\n"); s.push_str(" uint nthreads [[threads_per_grid]]) {\n"); - s.push_str(&persistent_prologue(CoreDialect::Metal)); + s.push_str(&persistent_prologue(CoreDialect::Metal, p.class.scratch_words_per_lane())); } else { s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); } @@ -534,7 +513,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: } Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&a, false))), Op::WLoad => format!("{d} = {d} ^ {};", fetch(word_index(&a, true))), - Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a), + Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a, p.class.scratch_slot_mask()), }; s.push_str(&format!(" {line} // {k}\n")); } @@ -599,7 +578,7 @@ fn cuda_instr_lines(p: &Program) -> String { Op::Load if load_width(ins) > 1 => wide_load_stmt(CoreDialect::Cuda, &d, &a, ins.width, WideSource::Stored, None), Op::Load => format!("{d} = {d} ^ ds[{a} & mask];"), Op::WLoad => format!("{d} = {d} ^ ds[(__shfl_sync(0xffffffffu, {a}, 0) & wmask) + lane];"), - Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a), + Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a, p.class.scratch_slot_mask()), }; s.push_str(&format!(" {line} // {k} {}\n", ins.op.name())); } @@ -671,7 +650,7 @@ pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String { let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" }; s.push_str(&format!("__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask{scratch_args}) {{\n")); if p.has_scratch() { - s.push_str(&persistent_prologue(CoreDialect::Cuda)); + s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane())); } else { s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); } @@ -792,7 +771,7 @@ pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" }; s.push_str(&format!("__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw{scratch_args}) {{\n")); if p.has_scratch() { - s.push_str(&persistent_prologue(CoreDialect::Cuda)); + s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane())); } else { s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); } @@ -878,7 +857,7 @@ fn opencl_instr_lines(p: &Program) -> String { Op::Load if load_width(ins) > 1 => wide_load_stmt(CoreDialect::OpenCl, &d, &a, ins.width, WideSource::Stored, None), Op::Load => format!("{d} = {d} ^ ds[{a} & mask];"), Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[(t_ & wmask) + lane]; }}"), - Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a), + Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a, p.class.scratch_slot_mask()), }; s.push_str(&format!(" {line} // {k} {}\n", ins.op.name())); } @@ -897,7 +876,7 @@ pub fn opencl_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" }; s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw{scratch_args}) {{\n")); if p.has_scratch() { - s.push_str(&persistent_prologue(CoreDialect::OpenCl)); + s.push_str(&persistent_prologue(CoreDialect::OpenCl, p.class.scratch_words_per_lane())); } else { s.push_str(" uint gid = (uint)get_global_id(0);\n"); } @@ -1029,7 +1008,7 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String { let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" }; s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask{scratch_args}) {{\n")); if p.has_scratch() { - s.push_str(&persistent_prologue(CoreDialect::OpenCl)); + s.push_str(&persistent_prologue(CoreDialect::OpenCl, p.class.scratch_words_per_lane())); } else { s.push_str(" uint gid = (uint)get_global_id(0);\n"); } @@ -1303,7 +1282,8 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!(" \"bytes_per_hash\": {},\n", p.bytes_per_hash())); if p.has_scratch() { s.push_str(&format!(" \"scratch_ops_per_hash\": {},\n", p.scratch_ops_per_hash())); - s.push_str(&format!(" \"scratch\": \"variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of {SCRATCH_SLOTS} 16-byte slots per lane (lane-major); slot = src & 0x{SCRATCH_SLOT_MASK:x}; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)\",\n")); + s.push_str(&format!(" \"scratch_kib_per_warp\": {},\n", p.class.scratch_kb)); + s.push_str(&format!(" \"scratch\": \"variant 5 (measurement only): persistent warps; a {kb} KiB scratch per warp of {slots} 16-byte slots per lane (lane-major); slot = src & 0x{smask:x}; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)\",\n", kb = p.class.scratch_kb, slots = p.class.scratch_slots_per_lane(), smask = p.class.scratch_slot_mask())); } s.push_str(&format!(" \"wide_load\": \"read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, {FOLD_ROT}) * 0x{FOLD_MUL:08x}) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots\",\n")); } diff --git a/igneum-pow/src/generator.rs b/igneum-pow/src/generator.rs index 0092818e9..bcd8bec25 100644 --- a/igneum-pow/src/generator.rs +++ b/igneum-pow/src/generator.rs @@ -170,50 +170,70 @@ pub const WIDTH_WORDS: [u8; 3] = [1, 4, 16]; pub struct LoadClass { pub mix: [u8; 3], pub load_slots: u8, - /// Variant 5: `Some(k)` gives the program a 1 MiB per-warp scratch (the kernels run persistent warps) and - /// turns `k` of the load slots into scratch read-modify-writes. `None` for every other class. + /// Variant 5: `Some(k)` gives the program a per-warp scratch (the kernels run persistent warps) and turns `k` + /// of the load slots into scratch read-modify-writes. `None` for every other class. pub scratch: Option, + /// Variant 5: the scratch per warp in KiB (32 or 128; the whole working set of a card at full occupancy must + /// stay under 6 GB, coordinator's cap of 5 October 2026). 0 for every other class. + pub scratch_kb: u8, } -/// Scratch geometry (variant 5): 2^11 slots of 16 bytes per lane (32 KiB), 32 lanes per warp (1 MiB), lane-major. -pub const SCRATCH_SLOT_BITS: u32 = 11; -pub const SCRATCH_SLOTS: usize = 1 << SCRATCH_SLOT_BITS; -pub const SCRATCH_SLOT_MASK: u32 = SCRATCH_SLOTS as u32 - 1; -pub const SCRATCH_WORDS_PER_LANE: usize = SCRATCH_SLOTS * 4; -pub const SCRATCH_BYTES_PER_WARP: usize = SCRATCH_WORDS_PER_LANE * 4 * LANES; +/// Scratch geometry (variant 5): 16-byte slots, lane-major, 32 lanes per warp; `scratch_kb` KiB per warp gives +/// `scratch_kb x 2` slots per lane (32 KiB: 64 slots, 128 KiB: 256 slots). +pub const SCRATCH_SLOT_BYTES: usize = 16; + +impl LoadClass { + /// Slots per lane of the scratch (0 without one). + pub fn scratch_slots_per_lane(&self) -> usize { + self.scratch_kb as usize * 1024 / LANES / SCRATCH_SLOT_BYTES + } + pub fn scratch_slot_mask(&self) -> u32 { + self.scratch_slots_per_lane().saturating_sub(1) as u32 + } + pub fn scratch_words_per_lane(&self) -> usize { + self.scratch_slots_per_lane() * 4 + } + pub fn scratch_bytes_per_warp(&self) -> usize { + self.scratch_kb as usize * 1024 + } +} impl LoadClass { /// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash. - pub const V2: LoadClass = LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None }; + pub const V2: LoadClass = LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0 }; /// A fixed width (1, 4 or 16 words) with `load_slots` loads per program. pub fn fixed(width_words: u8, load_slots: u8) -> LoadClass { let mut mix = [0u8; 3]; let i = WIDTH_WORDS.iter().position(|&w| w == width_words).expect("width must be 1, 4 or 16 words"); mix[i] = 100; - LoadClass { mix, load_slots, scratch: None } + LoadClass { mix, load_slots, scratch: None, scratch_kb: 0 } } /// Per-load width drawn from `mix` (percent for 4, 16, 64 bytes), 16 loads per program. pub fn mixed(mix: [u8; 3]) -> LoadClass { assert_eq!(mix.iter().map(|&m| m as u32).sum::(), 100, "the mix must sum to 100"); - LoadClass { mix, load_slots: LOAD_SLOTS as u8, scratch: None } + LoadClass { mix, load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0 } } - /// Variant 5: version 2 widths, 16 memory operations of which `k` are scratch read-modify-writes. - pub fn scratch(k: u8) -> LoadClass { + /// Variant 5: version 2 widths, 16 memory operations of which `k` are scratch read-modify-writes into a + /// scratch of `kb` KiB per warp (a power of two, 1 to 128: at least one slot per lane, under the 6 GB cap). + pub fn scratch(k: u8, kb: u8) -> LoadClass { assert!(k as usize <= LOAD_SLOTS); - LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: Some(k) } + assert!(kb.is_power_of_two() && kb <= 128, "scratch per warp must be a power of two up to 128 KiB"); + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: Some(k), scratch_kb: kb } } - /// Parse "p4,p16,p64" or one of the names of [`LoadClass::name`]. + /// Parse "p4,p16,p64" or one of the names of [`LoadClass::name`] ("scr4k32": 4 scratch ops, 32 KiB per warp). pub fn parse(s: &str) -> Option { - if let Some(k) = s.strip_prefix("scr") { + if let Some(rest) = s.strip_prefix("scr") { + let (k, kb) = rest.split_once('k')?; let k: u8 = k.parse().ok()?; - if k as usize > LOAD_SLOTS { + let kb: u8 = kb.parse().ok()?; + if k as usize > LOAD_SLOTS || !kb.is_power_of_two() || kb > 128 { return None; } - return Some(LoadClass::scratch(k)); + return Some(LoadClass::scratch(k, kb)); } let (mix_s, slots) = match s.split_once("x") { Some((m, n)) if !m.contains(',') => (m, n.parse::().ok()?), @@ -235,7 +255,7 @@ impl LoadClass { if slots == 0 || slots as usize >= INSTR_COUNT { return None; } - Some(LoadClass { mix, load_slots: slots, scratch: None }) + Some(LoadClass { mix, load_slots: slots, scratch: None, scratch_kb: 0 }) } /// Scratch read-modify-writes per program (0 without a scratch). @@ -253,7 +273,7 @@ impl LoadClass { return "v2".to_string(); } if let Some(k) = self.scratch { - return format!("scr{k}"); + return format!("scr{k}k{}", self.scratch_kb); } let base = match self.mix { [100, 0, 0] => "w4".to_string(), @@ -387,6 +407,7 @@ pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &L if let Some(k) = class.scratch { b.extend_from_slice(b"scratch/"); b.push(k); + b.push(class.scratch_kb); } fnv1a64(&b) } @@ -855,10 +876,12 @@ mod tests { assert!(ids.insert(p.program_id()), "{name}: program id collides"); } // variant 5: k scratch ops among the 16 memory operations, the rest one-word loads - for k in [0u8, 2, 4, 8] { - let c = LoadClass::parse(&format!("scr{k}")).unwrap(); - assert_eq!(c, LoadClass::scratch(k)); - assert_eq!(c.name(), format!("scr{k}")); + for (k, kb) in [(0u8, 32u8), (2, 32), (4, 128), (8, 128)] { + let c = LoadClass::parse(&format!("scr{k}k{kb}")).unwrap(); + assert_eq!(c, LoadClass::scratch(k, kb)); + assert_eq!(c.name(), format!("scr{k}k{kb}")); + assert_eq!(c.scratch_bytes_per_warp(), kb as usize * 1024); + assert_eq!(c.scratch_slots_per_lane(), kb as usize * 2); let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); assert_eq!(p.loads_per_hash(), 128); assert_eq!(p.scratch_ops_per_hash(), 8 * k as usize); @@ -866,7 +889,9 @@ mod tests { assert!(p.instrs.iter().all(|i| i.width == 1)); assert!(ids.insert(p.program_id()), "scr{k}: program id collides"); } - assert!(!LoadClass::scratch(0).is_v2()); + assert!(!LoadClass::scratch(0, 32).is_v2()); + assert_ne!(LoadClass::scratch(4, 32).name(), LoadClass::scratch(4, 128).name()); + assert_eq!(LoadClass::parse("scr4"), None); // a class with the version 2 widths but another slot count takes the extra roll: a different stream let w4x8 = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::fixed(1, 8)); assert_ne!(w4x8.instrs, v2.instrs); diff --git a/igneum-pow/src/verify.rs b/igneum-pow/src/verify.rs index 45808be94..fedfe1b91 100644 --- a/igneum-pow/src/verify.rs +++ b/igneum-pow/src/verify.rs @@ -1,9 +1,7 @@ //! The CPU reference interpreter for one 32-lane warp (`cpuWarpTraced` in the Swift) and the API the node //! calls. Dataset words come from the memory-hard cache (default) or from the closed form (old packs). -use crate::generator::{ - generate, generate_class, Instr, LoadClass, Op, Program, ITERATIONS, LANES, SCRATCH_SLOTS, SCRATCH_SLOT_MASK, -}; +use crate::generator::{generate, generate_class, Instr, LoadClass, Op, Program, ITERATIONS, LANES}; use crate::memhard::MemhardCpu; use crate::seed::day_key; @@ -45,9 +43,10 @@ pub fn scratch_rewrite(x: u32, w: &[u32; 3]) -> [u32; 3] { } /// The CPU model of one unit's scratch (variant 5): per lane, the written slots and their words. Unwritten slots -/// read as [`scratch_fill`]. A unit touches at most `scratch ops x 32` slots, so the model is small whatever the -/// nominal 1 MiB; a GPU keeps the real 1 MiB per resident warp with a per-unit tag per slot. +/// read as [`scratch_fill`]. A unit touches at most `scratch ops x 32` slots; a GPU keeps the real scratch per +/// resident warp with a per-unit tag per slot. pub struct ScratchModel { + slots: usize, written: Vec, data: Vec<[u32; 3]>, pub reads: usize, @@ -55,13 +54,19 @@ pub struct ScratchModel { } impl ScratchModel { - pub fn new() -> Self { - Self { written: vec![false; LANES * SCRATCH_SLOTS], data: vec![[0; 3]; LANES * SCRATCH_SLOTS], reads: 0, writes: 0 } + pub fn new(slots_per_lane: usize) -> Self { + Self { + slots: slots_per_lane, + written: vec![false; LANES * slots_per_lane], + data: vec![[0; 3]; LANES * slots_per_lane], + reads: 0, + writes: 0, + } } /// Read slot `slot` of `lane`, then rewrite it from the fold result `x`. Returns the three words read. #[inline] pub fn rmw(&mut self, seed: &[u32; 8], base: u32, lane: usize, slot: u32, dst: u32) -> u32 { - let i = lane * SCRATCH_SLOTS + slot as usize; + let i = lane * self.slots + slot as usize; let w = if self.written[i] { self.data[i] } else { @@ -80,12 +85,6 @@ impl ScratchModel { } } -impl Default for ScratchModel { - fn default() -> Self { - Self::new() - } -} - /// Dataset element, closed form of (day words, index). The original prototype's six-operation element. #[inline(always)] pub fn dataset_elem(i: u32, d0: u32, d1: u32) -> u32 { @@ -258,7 +257,8 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, let mut items_derived = 0usize; let mut idx = [0u32; LANES]; let mut val = [0u32; LANES]; - let mut scratch = if program.has_scratch() { Some(ScratchModel::new()) } else { None }; + let mut scratch = if program.has_scratch() { Some(ScratchModel::new(program.class.scratch_slots_per_lane())) } else { None }; + let slot_mask = program.class.scratch_slot_mask(); for _ in 0..ITERATIONS { let sel = r[0]; for ins in &program.instrs { @@ -267,7 +267,7 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, let m = scratch.as_mut().expect("a scratch op needs a scratch class"); let (d, a) = (ins.dst as usize, ins.src as usize); for lane in 0..LANES { - let slot = r[a][lane] & SCRATCH_SLOT_MASK; + let slot = r[a][lane] & slot_mask; r[d][lane] = m.rmw(&program.seed, base_nonce, lane, slot, r[d][lane]); } } @@ -538,14 +538,14 @@ mod tests { assert_eq!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 1)); assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2)); assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 64, 3, 100, 1)); - let mut m = ScratchModel::new(); + let mut m = ScratchModel::new(256); let w = [scratch_fill(&seed, 32, 3, 100, 0), scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2)]; let x = m.rmw(&seed, 32, 3, 100, 0xabcd); assert_eq!(x, fold_words(0xabcd, &w)); let x2 = m.rmw(&seed, 32, 3, 100, 0xabcd); assert_eq!(x2, fold_words(0xabcd, &scratch_rewrite(x, &w))); assert_eq!(m.reads, 2); - let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::scratch(4)); + let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::scratch(4, 128)); assert_eq!(e.program.scratch_ops_per_hash(), 32); assert_eq!(e.hash_warp(0), e.hash_warp(0)); } diff --git a/proto-cuda/nvrtc/packfile.h b/proto-cuda/nvrtc/packfile.h index fa407f93c..ba1719e8e 100644 --- a/proto-cuda/nvrtc/packfile.h +++ b/proto-cuda/nvrtc/packfile.h @@ -25,6 +25,9 @@ typedef struct { uint32_t datasetLog2, cacheLog2Words, cacheSegments, datasetMode, generator; uint32_t seedw[8], keyw[8]; char seedString[600]; + // read-width experiment (5 October 2026): the load class (0 when absent), bytes per hash, variant 5's scratch + uint32_t loadsPerHash, bytesPerHash, scratchOps, persistent, scratchWordsPerLane; + char loadClass[64]; // seeds.txt (or program.h): the seeds as the worker protocol carries them char epochHex[65]; char dayHex[PF_HEX_CAP]; @@ -254,6 +257,12 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) { if (!pf_define_u32(prog, "IGNEUM_CACHE_SEGMENTS", &pk->cacheSegments)) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_CACHE_SEGMENTS"); } if (!pf_define_u32(prog, "IGNEUM_GENERATOR", &pk->generator)) pk->generator = 1; if (pf_define_words(prog, "IGNEUM_SEEDW_INIT", pk->seedw, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_SEEDW_INIT with 8 words"); } + pk->loadsPerHash = 128; pf_define_u32(prog, "IGNEUM_LOADS_PER_HASH", &pk->loadsPerHash); + pk->bytesPerHash = pk->loadsPerHash * 4u; pf_define_u32(prog, "IGNEUM_BYTES_PER_HASH", &pk->bytesPerHash); + pk->scratchOps = 0; pf_define_u32(prog, "IGNEUM_SCRATCH_OPS", &pk->scratchOps); + pk->persistent = 0; pf_define_u32(prog, "IGNEUM_PERSISTENT_WARPS", &pk->persistent); + pk->scratchWordsPerLane = 8192; pf_define_u32(prog, "IGNEUM_SCRATCH_WORDS_PER_LANE", &pk->scratchWordsPerLane); + strcpy(pk->loadClass, "v2"); pf_define_str(prog, "IGNEUM_LOAD_CLASS", pk->loadClass, sizeof(pk->loadClass)); if (pf_define_words(prog, "IGNEUM_KEY_INIT", pk->keyw, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_KEY_INIT with 8 words"); } if (!pf_define_str(prog, "IGNEUM_SEED_STRING", pk->seedString, sizeof(pk->seedString))) strncpy(pk->seedString, "(no IGNEUM_SEED_STRING)", sizeof(pk->seedString) - 1); pf_define_str(prog, "IGNEUM_SEED_BYTES_HEX", ehex, sizeof(ehex)); diff --git a/proto-cuda/packs-readwidth/scr0/kernel.cl b/proto-cuda/packs-readwidth/scr0k32/kernel.cl similarity index 99% rename from proto-cuda/packs-readwidth/scr0/kernel.cl rename to proto-cuda/packs-readwidth/scr0k32/kernel.cl index 5ab647860..040d671a2 100644 --- a/proto-cuda/packs-readwidth/scr0/kernel.cl +++ b/proto-cuda/packs-readwidth/scr0k32/kernel.cl @@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n // One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the // lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and // __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; diff --git a/proto-cuda/packs-readwidth/scr0/kernel.cu b/proto-cuda/packs-readwidth/scr0k32/kernel.cu similarity index 98% rename from proto-cuda/packs-readwidth/scr0/kernel.cu rename to proto-cuda/packs-readwidth/scr0k32/kernel.cu index f7ac2c32d..14709616c 100644 --- a/proto-cuda/packs-readwidth/scr0/kernel.cu +++ b/proto-cuda/packs-readwidth/scr0k32/kernel.cu @@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem // One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every // __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a // 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. __device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; - uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { uint32_t gid = g_ * 32u + lane; uint32_t gbase = baseNonce + g_ * 32u; diff --git a/proto-cuda/packs-readwidth/scr0/kernel_bound.cl b/proto-cuda/packs-readwidth/scr0k32/kernel_bound.cl similarity index 99% rename from proto-cuda/packs-readwidth/scr0/kernel_bound.cl rename to proto-cuda/packs-readwidth/scr0k32/kernel_bound.cl index eaba1546c..057f228d0 100644 --- a/proto-cuda/packs-readwidth/scr0/kernel_bound.cl +++ b/proto-cuda/packs-readwidth/scr0k32/kernel_bound.cl @@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n // One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the // lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and // __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; diff --git a/proto-cuda/packs-readwidth/scr0/kernel_bound.cu b/proto-cuda/packs-readwidth/scr0k32/kernel_bound.cu similarity index 98% rename from proto-cuda/packs-readwidth/scr0/kernel_bound.cu rename to proto-cuda/packs-readwidth/scr0k32/kernel_bound.cu index 5f1b9aa92..47bd8b04d 100644 --- a/proto-cuda/packs-readwidth/scr0/kernel_bound.cu +++ b/proto-cuda/packs-readwidth/scr0k32/kernel_bound.cu @@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) { __device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } __device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. __device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; - uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { uint32_t gid = g_ * 32u + lane; uint32_t gbase = baseNonce + g_ * 32u; diff --git a/proto-cuda/packs-readwidth/scr0/memhard.h b/proto-cuda/packs-readwidth/scr0k32/memhard.h similarity index 100% rename from proto-cuda/packs-readwidth/scr0/memhard.h rename to proto-cuda/packs-readwidth/scr0k32/memhard.h diff --git a/proto-cuda/packs-readwidth/scr0/memhard.metal b/proto-cuda/packs-readwidth/scr0k32/memhard.metal similarity index 100% rename from proto-cuda/packs-readwidth/scr0/memhard.metal rename to proto-cuda/packs-readwidth/scr0k32/memhard.metal diff --git a/proto-cuda/packs-readwidth/scr0/program.h b/proto-cuda/packs-readwidth/scr0k32/program.h similarity index 91% rename from proto-cuda/packs-readwidth/scr0/program.h rename to proto-cuda/packs-readwidth/scr0k32/program.h index 2409c10b0..04c678b6e 100644 --- a/proto-cuda/packs-readwidth/scr0/program.h +++ b/proto-cuda/packs-readwidth/scr0k32/program.h @@ -15,7 +15,7 @@ #define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" #define IGNEUM_GENERATOR 2 #define IGNEUM_PROGRAM_ATTEMPT 0 -#define IGNEUM_PROGRAM_ID 0x2f098ae568f38029ull +#define IGNEUM_PROGRAM_ID 0xe0b70cd155c28f4bull #define IGNEUM_DAY_STRING "2026-10-03" #define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" #define IGNEUM_DAY0 0x3067619fu @@ -30,20 +30,20 @@ #define IGNEUM_OP_MIX "load=16 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 rotl=1" // Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads // the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x. -#define IGNEUM_LOAD_CLASS "scr0" +#define IGNEUM_LOAD_CLASS "scr0k32" #define IGNEUM_LOAD_SLOTS 16 #define IGNEUM_LOAD_MIX { 100, 0, 0 } #define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program #define IGNEUM_BYTES_PER_HASH 512 #define IGNEUM_FOLD_ROT 11 #define IGNEUM_FOLD_MUL 0x9e3779b1u -// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch, +// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch, // groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit). #define IGNEUM_PERSISTENT_WARPS 1 #define IGNEUM_SCRATCH_OPS 0 // scratch read-modify-writes per program (0 per hash) -#define IGNEUM_SCRATCH_SLOTS 2048u -#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u -#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u +#define IGNEUM_SCRATCH_SLOTS 64u +#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u +#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u // 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) #define IGNEUM_DATASET_MODE 1 diff --git a/proto-cuda/packs-readwidth/scr0/program.json b/proto-cuda/packs-readwidth/scr0k32/program.json similarity index 96% rename from proto-cuda/packs-readwidth/scr0/program.json rename to proto-cuda/packs-readwidth/scr0k32/program.json index ad5bdb8c0..2d28cab4a 100644 --- a/proto-cuda/packs-readwidth/scr0/program.json +++ b/proto-cuda/packs-readwidth/scr0k32/program.json @@ -2,7 +2,7 @@ "format": "igneum-program-pack-3", "generator": 2, "attempt": 0, - "program_id": "0x2f098ae568f38029", + "program_id": "0xe0b70cd155c28f4b", "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", "dataset_mode": "memory-hard", "seed": "igneum-genesis", @@ -15,13 +15,14 @@ "iterations": 8, "instruction_count": 64, "loads_per_hash": 128, - "load_class": "scr0", + "load_class": "scr0k32", "load_slots": 16, "load_mix_percent_4_16_64": [100, 0, 0], "load_width_counts_4_16_64": [16, 0, 0], "bytes_per_hash": 512, "scratch_ops_per_hash": 0, - "scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)", + "scratch_kib_per_warp": 32, + "scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)", "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", "op_mix": {"load": 16, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "rotl": 1}, "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", diff --git a/proto-cuda/packs-readwidth/scr0/program.metal b/proto-cuda/packs-readwidth/scr0k32/program.metal similarity index 99% rename from proto-cuda/packs-readwidth/scr0/program.metal rename to proto-cuda/packs-readwidth/scr0k32/program.metal index 9835a937b..a64d3b5ff 100644 --- a/proto-cuda/packs-readwidth/scr0/program.metal +++ b/proto-cuda/packs-readwidth/scr0k32/program.metal @@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) { return x; } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], uint lane = tid & 31u; uint warp_ = tid >> 5; uint nwarps_ = nthreads >> 5; - device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; diff --git a/proto-cuda/packs-readwidth/scr0/program_bound.metal b/proto-cuda/packs-readwidth/scr0k32/program_bound.metal similarity index 99% rename from proto-cuda/packs-readwidth/scr0/program_bound.metal rename to proto-cuda/packs-readwidth/scr0k32/program_bound.metal index fbfa7581b..9106aa169 100644 --- a/proto-cuda/packs-readwidth/scr0/program_bound.metal +++ b/proto-cuda/packs-readwidth/scr0k32/program_bound.metal @@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) { return x; } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], uint lane = tid & 31u; uint warp_ = tid >> 5; uint nwarps_ = nthreads >> 5; - device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; diff --git a/proto-cuda/packs-readwidth/scr0/vectors.h b/proto-cuda/packs-readwidth/scr0k32/vectors.h similarity index 100% rename from proto-cuda/packs-readwidth/scr0/vectors.h rename to proto-cuda/packs-readwidth/scr0k32/vectors.h diff --git a/proto-cuda/packs-readwidth/scr0/vectors.json b/proto-cuda/packs-readwidth/scr0k32/vectors.json similarity index 100% rename from proto-cuda/packs-readwidth/scr0/vectors.json rename to proto-cuda/packs-readwidth/scr0k32/vectors.json diff --git a/proto-cuda/packs-readwidth/scr2/kernel.cl b/proto-cuda/packs-readwidth/scr2k128/kernel.cl similarity index 93% rename from proto-cuda/packs-readwidth/scr2/kernel.cl rename to proto-cuda/packs-readwidth/scr2k128/kernel.cl index b7d17259d..8fc591da5 100644 --- a/proto-cuda/packs-readwidth/scr2/kernel.cl +++ b/proto-cuda/packs-readwidth/scr2k128/kernel.cl @@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n // One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the // lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and // __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + { uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl r4 = r0 * r2 + r4; // 19 mad @@ -244,7 +244,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r3 = r3 ^ ds[r1 & mask]; // 31 load r1 = r1 ^ ds[r0 & mask]; // 32 load r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor r4 = r4 ^ ds[r0 & mask]; // 37 load diff --git a/proto-cuda/packs-readwidth/scr2/kernel.cu b/proto-cuda/packs-readwidth/scr2k128/kernel.cu similarity index 88% rename from proto-cuda/packs-readwidth/scr2/kernel.cu rename to proto-cuda/packs-readwidth/scr2k128/kernel.cu index b1b566dea..6d8bf993d 100644 --- a/proto-cuda/packs-readwidth/scr2/kernel.cu +++ b/proto-cuda/packs-readwidth/scr2k128/kernel.cu @@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem // One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every // __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a // 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. __device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; - uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { uint32_t gid = g_ * 32u + lane; uint32_t gbase = baseNonce + g_ * 32u; @@ -86,7 +86,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + { uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl r4 = r0 * r2 + r4; // 19 mad @@ -104,7 +104,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc r3 = r3 ^ ds[r1 & mask]; // 31 load r1 = r1 ^ ds[r0 & mask]; // 32 load r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + { uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor r4 = r4 ^ ds[r0 & mask]; // 37 load diff --git a/proto-cuda/packs-readwidth/scr2/kernel_bound.cl b/proto-cuda/packs-readwidth/scr2k128/kernel_bound.cl similarity index 90% rename from proto-cuda/packs-readwidth/scr2/kernel_bound.cl rename to proto-cuda/packs-readwidth/scr2k128/kernel_bound.cl index 9889162fc..b08bc77a8 100644 --- a/proto-cuda/packs-readwidth/scr2/kernel_bound.cl +++ b/proto-cuda/packs-readwidth/scr2k128/kernel_bound.cl @@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n // One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the // lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and // __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + { uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl r4 = r0 * r2 + r4; // 19 mad @@ -244,7 +244,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r3 = r3 ^ ds[r1 & mask]; // 31 load r1 = r1 ^ ds[r0 & mask]; // 32 load r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor r4 = r4 ^ ds[r0 & mask]; // 37 load @@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -337,7 +337,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + { uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl r4 = r0 * r2 + r4; // 19 mad @@ -355,7 +355,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon r3 = r3 ^ ds[r1 & mask]; // 31 load r1 = r1 ^ ds[r0 & mask]; // 32 load r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor r4 = r4 ^ ds[r0 & mask]; // 37 load diff --git a/proto-cuda/packs-readwidth/scr2/kernel_bound.cu b/proto-cuda/packs-readwidth/scr2k128/kernel_bound.cu similarity index 85% rename from proto-cuda/packs-readwidth/scr2/kernel_bound.cu rename to proto-cuda/packs-readwidth/scr2k128/kernel_bound.cu index b012ad62e..81875c255 100644 --- a/proto-cuda/packs-readwidth/scr2/kernel_bound.cu +++ b/proto-cuda/packs-readwidth/scr2k128/kernel_bound.cu @@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) { __device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } __device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. __device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; - uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { uint32_t gid = g_ * 32u + lane; uint32_t gbase = baseNonce + g_ * 32u; @@ -62,7 +62,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + { uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl r4 = r0 * r2 + r4; // 19 mad @@ -80,7 +80,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba r3 = r3 ^ ds[r1 & mask]; // 31 load r1 = r1 ^ ds[r0 & mask]; // 32 load r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + { uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor r4 = r4 ^ ds[r0 & mask]; // 37 load diff --git a/proto-cuda/packs-readwidth/scr2/memhard.h b/proto-cuda/packs-readwidth/scr2k128/memhard.h similarity index 100% rename from proto-cuda/packs-readwidth/scr2/memhard.h rename to proto-cuda/packs-readwidth/scr2k128/memhard.h diff --git a/proto-cuda/packs-readwidth/scr2/memhard.metal b/proto-cuda/packs-readwidth/scr2k128/memhard.metal similarity index 100% rename from proto-cuda/packs-readwidth/scr2/memhard.metal rename to proto-cuda/packs-readwidth/scr2k128/memhard.metal diff --git a/proto-cuda/packs-readwidth/scr2k128/program.h b/proto-cuda/packs-readwidth/scr2k128/program.h new file mode 100644 index 000000000..b2b831632 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr2k128/program.h @@ -0,0 +1,67 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Program metadata for host.cu plus the launch wrappers defined in kernel.cu. +// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#ifndef IGNEUM_NO_CUDA +#include +#endif + +#define IGNEUM_SEED_STRING "igneum-genesis" +#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" +#define IGNEUM_GENERATOR 2 +#define IGNEUM_PROGRAM_ATTEMPT 0 +#define IGNEUM_PROGRAM_ID 0xe0afe0d155bc25d9ull +#define IGNEUM_DAY_STRING "2026-10-03" +#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" +#define IGNEUM_DAY0 0x3067619fu +#define IGNEUM_DAY1 0x3c269176u +#define IGNEUM_DATASET_LOG2 28 +#define IGNEUM_MASK 0x0fffffffu +#define IGNEUM_LANES 32 +#define IGNEUM_ITERATIONS 8 +#define IGNEUM_INSTR_COUNT 64 +#define IGNEUM_LOADS_PER_HASH 128 +#define IGNEUM_WIDE_LOADS_PER_HASH 0 +#define IGNEUM_OP_MIX "load=14 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 scratch=2 rotl=1" +// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads +// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x. +#define IGNEUM_LOAD_CLASS "scr2k128" +#define IGNEUM_LOAD_SLOTS 16 +#define IGNEUM_LOAD_MIX { 100, 0, 0 } +#define IGNEUM_LOAD_WIDTH_COUNTS { 14, 0, 0 } // loads of 4, 16, 64 bytes per program +#define IGNEUM_BYTES_PER_HASH 448 +#define IGNEUM_FOLD_ROT 11 +#define IGNEUM_FOLD_MUL 0x9e3779b1u +// Variant 5: persistent warps, a 128 KiB scratch per launched warp (the host launches N warps and passes scratch, +// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit). +#define IGNEUM_PERSISTENT_WARPS 1 +#define IGNEUM_SCRATCH_OPS 2 // scratch read-modify-writes per program (16 per hash) +#define IGNEUM_SCRATCH_SLOTS 256u +#define IGNEUM_SCRATCH_WORDS_PER_LANE 1024u +#define IGNEUM_SCRATCH_BYTES_PER_WARP 131072u +// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) +#define IGNEUM_DATASET_MODE 1 + +#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u } +#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u } +#define IGNEUM_CACHE_LOG2_WORDS 26 +#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6 +#define IGNEUM_CACHE_SEGMENTS 65536u +#define IGNEUM_ITEM_ROUNDS 8 +#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u } +#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u } +#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u } + +#ifndef IGNEUM_NO_CUDA +// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError(). +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt); +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#endif diff --git a/proto-cuda/packs-readwidth/scr2/program.json b/proto-cuda/packs-readwidth/scr2k128/program.json similarity index 98% rename from proto-cuda/packs-readwidth/scr2/program.json rename to proto-cuda/packs-readwidth/scr2k128/program.json index b9b6b935c..4b6d69fa5 100644 --- a/proto-cuda/packs-readwidth/scr2/program.json +++ b/proto-cuda/packs-readwidth/scr2k128/program.json @@ -2,7 +2,7 @@ "format": "igneum-program-pack-3", "generator": 2, "attempt": 0, - "program_id": "0x2f0988e568f37cc3", + "program_id": "0xe0afe0d155bc25d9", "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", "dataset_mode": "memory-hard", "seed": "igneum-genesis", @@ -15,13 +15,14 @@ "iterations": 8, "instruction_count": 64, "loads_per_hash": 128, - "load_class": "scr2", + "load_class": "scr2k128", "load_slots": 16, "load_mix_percent_4_16_64": [100, 0, 0], "load_width_counts_4_16_64": [14, 0, 0], "bytes_per_hash": 448, "scratch_ops_per_hash": 16, - "scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)", + "scratch_kib_per_warp": 128, + "scratch": "variant 5 (measurement only): persistent warps; a 128 KiB scratch per warp of 256 16-byte slots per lane (lane-major); slot = src & 0xff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)", "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", "op_mix": {"load": 14, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "scratch": 2, "rotl": 1}, "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", diff --git a/proto-cuda/packs-readwidth/scr2/program.metal b/proto-cuda/packs-readwidth/scr2k128/program.metal similarity index 83% rename from proto-cuda/packs-readwidth/scr2/program.metal rename to proto-cuda/packs-readwidth/scr2k128/program.metal index 5c8835ec2..cf5d87bc2 100644 --- a/proto-cuda/packs-readwidth/scr2/program.metal +++ b/proto-cuda/packs-readwidth/scr2k128/program.metal @@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) { return x; } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], uint lane = tid & 31u; uint warp_ = tid >> 5; uint nwarps_ = nthreads >> 5; - device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -70,7 +70,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], r2 = r2 * r5; // 13 r1 = r1 ^ dataset[r2 & MASK]; // 14 r7 = rotl_imm(r7, 1u); // 15 - { uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + { uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 r7 = r7 ^ dataset[r4 & MASK]; // 17 r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 r4 = r0 * r2 + r4; // 19 @@ -88,7 +88,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], r3 = r3 ^ dataset[r1 & MASK]; // 31 r1 = r1 ^ dataset[r0 & MASK]; // 32 r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 - { uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + { uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 r0 = r0 * r3; // 35 r2 = r2 ^ r5; // 36 r4 = r4 ^ dataset[r0 & MASK]; // 37 diff --git a/proto-cuda/packs-readwidth/scr2/program_bound.metal b/proto-cuda/packs-readwidth/scr2k128/program_bound.metal similarity index 84% rename from proto-cuda/packs-readwidth/scr2/program_bound.metal rename to proto-cuda/packs-readwidth/scr2k128/program_bound.metal index b521473ef..ba42f27c1 100644 --- a/proto-cuda/packs-readwidth/scr2/program_bound.metal +++ b/proto-cuda/packs-readwidth/scr2k128/program_bound.metal @@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) { return x; } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], uint lane = tid & 31u; uint warp_ = tid >> 5; uint nwarps_ = nthreads >> 5; - device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -72,7 +72,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], r2 = r2 * r5; // 13 r1 = r1 ^ dataset[r2 & MASK]; // 14 r7 = rotl_imm(r7, 1u); // 15 - { uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + { uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 r7 = r7 ^ dataset[r4 & MASK]; // 17 r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 r4 = r0 * r2 + r4; // 19 @@ -90,7 +90,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], r3 = r3 ^ dataset[r1 & MASK]; // 31 r1 = r1 ^ dataset[r0 & MASK]; // 32 r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 - { uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + { uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 r0 = r0 * r3; // 35 r2 = r2 ^ r5; // 36 r4 = r4 ^ dataset[r0 & MASK]; // 37 diff --git a/proto-cuda/packs-readwidth/scr4/vectors.h b/proto-cuda/packs-readwidth/scr2k128/vectors.h similarity index 60% rename from proto-cuda/packs-readwidth/scr4/vectors.h rename to proto-cuda/packs-readwidth/scr2k128/vectors.h index 88076f2e3..2a556f269 100644 --- a/proto-cuda/packs-readwidth/scr4/vectors.h +++ b/proto-cuda/packs-readwidth/scr2k128/vectors.h @@ -11,22 +11,22 @@ static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { { // base nonce 0 - 0x66cffcc97c46e625ull, 0xbc9019f8df50fbfdull, 0x65629c90dde6016eull, 0xd647a41effa03d3bull, 0x86da3b6bbd751b99ull, 0x6ccf4240a0fb2d19ull, 0xeb39a1e06f17378cull, 0x2ea6b349b289fb10ull, - 0x6067211e6c220500ull, 0x6e6095dfedd1360full, 0xbd1190d8b50e1b48ull, 0x216dc72a0c08d5b5ull, 0x5be1f8c080836b0cull, 0x2a32932a5953ed73ull, 0xcc2a3d68be83c802ull, 0xe46daca15338278full, - 0xb2de43b96761e459ull, 0x9004acd06588cbeaull, 0x6a9a3543cf93004full, 0xff956d859cb6e408ull, 0x4397ec6e3c5fb045ull, 0x521dea569cd481d5ull, 0x89832b34108759f0ull, 0xf66e393836ffe4eaull, - 0xb4e39af6c40ea2f4ull, 0x3adc22085dd8d648ull, 0x27efe270958bbfbbull, 0x6c80be0e8dca60d8ull, 0xa0afbc6a60260d59ull, 0x5d9a257fb9189537ull, 0xeb837aeef55dc3edull, 0xd381174dc14f8951ull + 0x870ae6d97d9e85d8ull, 0x82f91989add778d3ull, 0x127e79aa060861e3ull, 0x148dec51aec6ee46ull, 0xcb5b4b144055bebaull, 0x832ea2d8305f7177ull, 0x374d405f6f35141dull, 0xe83f0b25fcc5b98cull, + 0x8c6696ba39a3dfbcull, 0x5e588ceed27c0f20ull, 0x87382ec243a8208full, 0x831da1dd50dedfd7ull, 0x11af1b35d85d23faull, 0x3f203e64b5a6ae40ull, 0x5562b30447cf8941ull, 0xdbc5ddc8b06cab3cull, + 0x089351d365256721ull, 0xf4e67e3e0ce8ce3dull, 0x9a6581aba1e08812ull, 0xf8c1f2017f31d0b2ull, 0x00e282fb6d67ed1bull, 0x64ceb85ec97d5368ull, 0x0db3762c566cd35full, 0x9ecf6fb65b27a141ull, + 0xdfb6edb29d069ef0ull, 0xa3a2eb24fa67fb93ull, 0x2527bac1b676b544ull, 0x275704b557b5c1d5ull, 0x909fe0fbbb3d7e3eull, 0x6d3b7083f0b9dac1ull, 0xf7a58d446af0c3f0ull, 0x45a10eec5d0361ebull }, { // base nonce 4096 - 0xfb1f61aaeaeaef94ull, 0x184c2160963d8b57ull, 0x42a053c628625778ull, 0xeaad0e41c4770812ull, 0x1d6d389ceb462ce1ull, 0x4a639827672bdbd4ull, 0x857e42aa5a42dd6full, 0xb4ef399e5339979cull, - 0x497a29225b099233ull, 0x71d8b42862d81954ull, 0x0af995663313bf04ull, 0xf436fd126619d7a1ull, 0x199e4e3333cff269ull, 0x64077952f3775768ull, 0x51af1d126c5e8388ull, 0xffbaf44fe6b15cfdull, - 0xfc8fed86ecae34a7ull, 0x4cb548616f7a7d6bull, 0xc21d938c8b5bef35ull, 0x34789cbdd7088f71ull, 0xacb099a2c207d891ull, 0xfe1902d162374413ull, 0x26f7831c28f4020bull, 0xdf5192952b4af6b0ull, - 0xecab61fe88dbaff4ull, 0x941c491f7fdb86e5ull, 0x2b1900c53f746e77ull, 0x8c40507b1caffeb2ull, 0x7532a1ec2b9169efull, 0x1cf399b0c8bfb520ull, 0xdf003d2bb8a2cc0cull, 0x4da853307fc977a9ull + 0x5ba9a19be2ac506full, 0xf01aba1b9e1fbd4bull, 0x82576a1ada6a06aaull, 0x3cdfb035063961adull, 0x3b1c0146bee5cc0bull, 0xb9eb92e4388bb2edull, 0xfb3d93c5214abf98ull, 0xb624a0986997e24aull, + 0x6dc096da5e72a34aull, 0x598baa91443c82dbull, 0x689d8cef7afc8df5ull, 0x41ae32225004d576ull, 0x06adbced85e30f4dull, 0x1bef955028e11da8ull, 0x8e3f1fde00391e41ull, 0xf1e29599bd9c776eull, + 0xd85a42863321dfbcull, 0x2f220e8389179830ull, 0x658f1c559f2e3e28ull, 0x7ddc9adae1172cfaull, 0x493dad4ec6a7d467ull, 0x4f8cf35bbfa01901ull, 0x87971b6666cd0093ull, 0xbe71aa56ca4f56b0ull, + 0xe9cee6ec93582a96ull, 0xa10e3cc76912027cull, 0x0d3b49bb042a2033ull, 0x680fa00b5d161278ull, 0xdef83e00736c287aull, 0x354ed038aae23286ull, 0xd280ce9bd9a71c97ull, 0x4f48b2b22716cd62ull }, { // base nonce 1000000 - 0x3d094bd04694b96full, 0xaecbd76cecd1a20aull, 0xbcb86febe56b17feull, 0x98082b557ba97517ull, 0xbb5f94108888564bull, 0xea3284877a30fc87ull, 0xc608fa4d5a8bb2adull, 0x946c721c511e0729ull, - 0x46c6eba292083aedull, 0x936cb97231eb6795ull, 0xb1413c434c712cbbull, 0xedfd554d3948c1bdull, 0xa8a20cbef2faccd5ull, 0x5fe39d756cadbcadull, 0x208b2627380791feull, 0xf52f9374ce480218ull, - 0xb9db7cd8814eb29eull, 0xf32ed2192b5a8719ull, 0x4f1b06a054940aefull, 0x406df498e4365eb5ull, 0x1982075caad345efull, 0x590f725623dbbbbdull, 0xa26d9192dedfefa5ull, 0x36219ec00da18980ull, - 0x5361d0dcb0f8b1a3ull, 0x35bdefa2fbb5ffc3ull, 0xba4c2a4e473a9c80ull, 0x107d9d3030f8b9d3ull, 0xa8bb094266d6b987ull, 0x86164fdfbb1426e8ull, 0xa6e8cb895021cbbdull, 0xfe12809e9d99a243ull + 0x3220aa9dc0bca592ull, 0x5409a7301dfcc3b6ull, 0x31c9b27aad8ef845ull, 0x564ceb4647002ad9ull, 0xcb7dc5fb129b07a6ull, 0xd940b224720c0393ull, 0x0e7f07a345108096ull, 0xc55a7d439b46f6a4ull, + 0x0a0fab6343d5756aull, 0x397b5e178eeaa82cull, 0xe6a0adaa085838f8ull, 0x7389e3a61a07941bull, 0xeff9f511b46237bbull, 0xb35b6bf8e31ad14cull, 0x0b7d29ad2f411c1bull, 0x988e30319baf21daull, + 0x362ff74de128b05bull, 0x312357fc7857b317ull, 0xb006adf1d322c446ull, 0x4dcdf54a98b429eeull, 0x18a48dc7b38023bdull, 0xb2af25e04714b6ceull, 0x4d7b97fe2f335dc2ull, 0xc161c509e57b4616ull, + 0xb2b90c940eb64229ull, 0x2549b669b9e63f78ull, 0x1a1a4ff0e629079cull, 0x80ccafd83d359a54ull, 0x0232c0dfa9240d4dull, 0xc98a751590860b3full, 0x8e96a0aae858e53bull, 0x906b462113b3f109ull } }; diff --git a/proto-cuda/packs-readwidth/scr8/vectors.json b/proto-cuda/packs-readwidth/scr2k128/vectors.json similarity index 65% rename from proto-cuda/packs-readwidth/scr8/vectors.json rename to proto-cuda/packs-readwidth/scr2k128/vectors.json index 64e628d6e..cf0908113 100644 --- a/proto-cuda/packs-readwidth/scr8/vectors.json +++ b/proto-cuda/packs-readwidth/scr2k128/vectors.json @@ -8,22 +8,22 @@ "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", "warps": [ {"base_nonce": 0, "expected": [ - "0xd0f846c3cd57ae09", "0x4d5e1caf41761a5c", "0x7fc77bb7fa221a51", "0xe45b6317d4f52ae7", "0x09704a39107a3150", "0xd5166620f9dc49ab", "0xeaf1fa69e4075e55", "0x38e23d449b7616ae", - "0xf4593424c8427320", "0x9359a44a5149bd60", "0x2df93b5bc274f083", "0xfd80eb32016f8659", "0xdeb10b8a37fc3ee7", "0xc18e2ed77ab77f34", "0x7f39b618cc81c1b4", "0xdc1b7299961ad5ab", - "0x8f9c03d151013ec4", "0x668927a4a267d76f", "0x233a50c329caf635", "0x7c80406541ae3f15", "0x18fb38606c48af4a", "0x0452a913df5f112b", "0x59065cb670ba6d7d", "0xa2998e2ccd726df8", - "0xef690e221af37893", "0x4e87968e5c45903e", "0x6c6e5d4c1b35d7a9", "0x54d7e322426dcae3", "0x0d3ed6c08deda320", "0x3d6a4301fbb06a5a", "0x9eb78b04d2366566", "0x7806f8c64d2d09ff" + "0x870ae6d97d9e85d8", "0x82f91989add778d3", "0x127e79aa060861e3", "0x148dec51aec6ee46", "0xcb5b4b144055beba", "0x832ea2d8305f7177", "0x374d405f6f35141d", "0xe83f0b25fcc5b98c", + "0x8c6696ba39a3dfbc", "0x5e588ceed27c0f20", "0x87382ec243a8208f", "0x831da1dd50dedfd7", "0x11af1b35d85d23fa", "0x3f203e64b5a6ae40", "0x5562b30447cf8941", "0xdbc5ddc8b06cab3c", + "0x089351d365256721", "0xf4e67e3e0ce8ce3d", "0x9a6581aba1e08812", "0xf8c1f2017f31d0b2", "0x00e282fb6d67ed1b", "0x64ceb85ec97d5368", "0x0db3762c566cd35f", "0x9ecf6fb65b27a141", + "0xdfb6edb29d069ef0", "0xa3a2eb24fa67fb93", "0x2527bac1b676b544", "0x275704b557b5c1d5", "0x909fe0fbbb3d7e3e", "0x6d3b7083f0b9dac1", "0xf7a58d446af0c3f0", "0x45a10eec5d0361eb" ]}, {"base_nonce": 4096, "expected": [ - "0xe2c97c384a85c687", "0xcf905b005ab01ebc", "0x4526799fae210c6b", "0xd4daf72ed75e8a16", "0xb3703a9e7c6820a8", "0xb6be9393fbb920bd", "0xe2fff04fb7816eb8", "0xed76ad5dbf400f91", - "0xea7d4b58ab0eb8d1", "0xf67a73b01f030532", "0x35ab823037510099", "0x5de1b1b0b3a26c69", "0xbe5dd90dbb7e632b", "0x1475918a9237e24d", "0x177fd53c45634d71", "0xa7cf00759ba28ce0", - "0xe51be0584ac3fbb4", "0x427049cc778aab35", "0x826bab125577d172", "0xd705b891b16237f5", "0xdf622fd44b180a87", "0x359398ecb79ec2de", "0x2a1e075fb078da66", "0xd480ddd8e66d26e8", - "0x07e3e86f517de466", "0xa98b2f7423557445", "0x6bab15b36bb142fe", "0xf87d147bf2cc5c0b", "0xf0294ea2b2820e03", "0xf219ac95e823d794", "0x9fa3fea85bc54264", "0xc4af3fafd2ff5201" + "0x5ba9a19be2ac506f", "0xf01aba1b9e1fbd4b", "0x82576a1ada6a06aa", "0x3cdfb035063961ad", "0x3b1c0146bee5cc0b", "0xb9eb92e4388bb2ed", "0xfb3d93c5214abf98", "0xb624a0986997e24a", + "0x6dc096da5e72a34a", "0x598baa91443c82db", "0x689d8cef7afc8df5", "0x41ae32225004d576", "0x06adbced85e30f4d", "0x1bef955028e11da8", "0x8e3f1fde00391e41", "0xf1e29599bd9c776e", + "0xd85a42863321dfbc", "0x2f220e8389179830", "0x658f1c559f2e3e28", "0x7ddc9adae1172cfa", "0x493dad4ec6a7d467", "0x4f8cf35bbfa01901", "0x87971b6666cd0093", "0xbe71aa56ca4f56b0", + "0xe9cee6ec93582a96", "0xa10e3cc76912027c", "0x0d3b49bb042a2033", "0x680fa00b5d161278", "0xdef83e00736c287a", "0x354ed038aae23286", "0xd280ce9bd9a71c97", "0x4f48b2b22716cd62" ]}, {"base_nonce": 1000000, "expected": [ - "0x501f772483fac0a3", "0x461363da2c1539e0", "0x750050674de592af", "0x14ed105042cd912c", "0xc6477310878614eb", "0xe916b44e32e90e56", "0x531c92e69c2bdd73", "0x5bb0129ef7c3bd51", - "0x967ed91f7cbc7c79", "0x06fcc6a895b58b4a", "0x3bdb29fbf93cbff3", "0xbed385172d2e6abe", "0x921fc99ff4efac5e", "0x6b0090bedd9f69c7", "0x0b105c18aaa53ab0", "0xf0c4203561f3b94d", - "0x64abe33adccf7807", "0xd8f3a3b7e0242c04", "0x397da462f69fac5b", "0xe6b6b32e467b71be", "0x5318ac9e56278d04", "0xf7c5e348a4e1f5db", "0x79c994e6109646df", "0x9cdc42b0f6e82231", - "0x3f1160f02fd96ac7", "0xa69a23fc7058be21", "0xccde0a19bfdc25b6", "0xd3684a5b966f7497", "0x929db00b97a624f9", "0xfd14a40882d395a6", "0x49b0c1ecc514d6ad", "0x95d994c8349be12d" + "0x3220aa9dc0bca592", "0x5409a7301dfcc3b6", "0x31c9b27aad8ef845", "0x564ceb4647002ad9", "0xcb7dc5fb129b07a6", "0xd940b224720c0393", "0x0e7f07a345108096", "0xc55a7d439b46f6a4", + "0x0a0fab6343d5756a", "0x397b5e178eeaa82c", "0xe6a0adaa085838f8", "0x7389e3a61a07941b", "0xeff9f511b46237bb", "0xb35b6bf8e31ad14c", "0x0b7d29ad2f411c1b", "0x988e30319baf21da", + "0x362ff74de128b05b", "0x312357fc7857b317", "0xb006adf1d322c446", "0x4dcdf54a98b429ee", "0x18a48dc7b38023bd", "0xb2af25e04714b6ce", "0x4d7b97fe2f335dc2", "0xc161c509e57b4616", + "0xb2b90c940eb64229", "0x2549b669b9e63f78", "0x1a1a4ff0e629079c", "0x80ccafd83d359a54", "0x0232c0dfa9240d4d", "0xc98a751590860b3f", "0x8e96a0aae858e53b", "0x906b462113b3f109" ]} ], "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], diff --git a/proto-cuda/packs-readwidth/scr4/kernel.cl b/proto-cuda/packs-readwidth/scr2k32/kernel.cl similarity index 88% rename from proto-cuda/packs-readwidth/scr4/kernel.cl rename to proto-cuda/packs-readwidth/scr2k32/kernel.cl index a75793f6f..6befdde06 100644 --- a/proto-cuda/packs-readwidth/scr4/kernel.cl +++ b/proto-cuda/packs-readwidth/scr2k32/kernel.cl @@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n // One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the // lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and // __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + { uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl r4 = r0 * r2 + r4; // 19 mad @@ -242,9 +242,9 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add r6 = rotr_var(r6, r7); // 30 rotr r3 = r3 ^ ds[r1 & mask]; // 31 load - { uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r1 = r1 ^ ds[r0 & mask]; // 32 load r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor r4 = r4 ^ ds[r0 & mask]; // 37 load @@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r2 = r2 ^ ds[r7 & mask]; // 56 load r5 = r5 - r6; // 57 sub r1 = r1 ^ ds[r3 & mask]; // 58 load - { uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r1 = r1 ^ ds[r4 & mask]; // 59 load r4 = r4 - r6; // 60 sub r1 = r1 * r2; // 61 mul r3 = r6 * r0 + r3; // 62 mad diff --git a/proto-cuda/packs-readwidth/scr4/kernel.cu b/proto-cuda/packs-readwidth/scr2k32/kernel.cu similarity index 79% rename from proto-cuda/packs-readwidth/scr4/kernel.cu rename to proto-cuda/packs-readwidth/scr2k32/kernel.cu index fd9584208..5fd93b513 100644 --- a/proto-cuda/packs-readwidth/scr4/kernel.cu +++ b/proto-cuda/packs-readwidth/scr2k32/kernel.cu @@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem // One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every // __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a // 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. __device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; - uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { uint32_t gid = g_ * 32u + lane; uint32_t gbase = baseNonce + g_ * 32u; @@ -86,7 +86,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + { uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl r4 = r0 * r2 + r4; // 19 mad @@ -102,9 +102,9 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add r6 = rotr_var(r6, r7); // 30 rotr r3 = r3 ^ ds[r1 & mask]; // 31 load - { uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch + r1 = r1 ^ ds[r0 & mask]; // 32 load r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + { uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor r4 = r4 ^ ds[r0 & mask]; // 37 load @@ -129,7 +129,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc r2 = r2 ^ ds[r7 & mask]; // 56 load r5 = r5 - r6; // 57 sub r1 = r1 ^ ds[r3 & mask]; // 58 load - { uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch + r1 = r1 ^ ds[r4 & mask]; // 59 load r4 = r4 - r6; // 60 sub r1 = r1 * r2; // 61 mul r3 = r6 * r0 + r3; // 62 mad diff --git a/proto-cuda/packs-readwidth/scr4/kernel_bound.cl b/proto-cuda/packs-readwidth/scr2k32/kernel_bound.cl similarity index 83% rename from proto-cuda/packs-readwidth/scr4/kernel_bound.cl rename to proto-cuda/packs-readwidth/scr2k32/kernel_bound.cl index e33afd4bc..9fb87c712 100644 --- a/proto-cuda/packs-readwidth/scr4/kernel_bound.cl +++ b/proto-cuda/packs-readwidth/scr2k32/kernel_bound.cl @@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n // One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the // lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and // __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + { uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl r4 = r0 * r2 + r4; // 19 mad @@ -242,9 +242,9 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add r6 = rotr_var(r6, r7); // 30 rotr r3 = r3 ^ ds[r1 & mask]; // 31 load - { uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r1 = r1 ^ ds[r0 & mask]; // 32 load r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor r4 = r4 ^ ds[r0 & mask]; // 37 load @@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r2 = r2 ^ ds[r7 & mask]; // 56 load r5 = r5 - r6; // 57 sub r1 = r1 ^ ds[r3 & mask]; // 58 load - { uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r1 = r1 ^ ds[r4 & mask]; // 59 load r4 = r4 - r6; // 60 sub r1 = r1 * r2; // 61 mul r3 = r6 * r0 + r3; // 62 mad @@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -337,7 +337,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + { uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl r4 = r0 * r2 + r4; // 19 mad @@ -353,9 +353,9 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add r6 = rotr_var(r6, r7); // 30 rotr r3 = r3 ^ ds[r1 & mask]; // 31 load - { uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r1 = r1 ^ ds[r0 & mask]; // 32 load r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor r4 = r4 ^ ds[r0 & mask]; // 37 load @@ -380,7 +380,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon r2 = r2 ^ ds[r7 & mask]; // 56 load r5 = r5 - r6; // 57 sub r1 = r1 ^ ds[r3 & mask]; // 58 load - { uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r1 = r1 ^ ds[r4 & mask]; // 59 load r4 = r4 - r6; // 60 sub r1 = r1 * r2; // 61 mul r3 = r6 * r0 + r3; // 62 mad diff --git a/proto-cuda/packs-readwidth/scr4/kernel_bound.cu b/proto-cuda/packs-readwidth/scr2k32/kernel_bound.cu similarity index 75% rename from proto-cuda/packs-readwidth/scr4/kernel_bound.cu rename to proto-cuda/packs-readwidth/scr2k32/kernel_bound.cu index a4d184a0a..94e353cf8 100644 --- a/proto-cuda/packs-readwidth/scr4/kernel_bound.cu +++ b/proto-cuda/packs-readwidth/scr2k32/kernel_bound.cu @@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) { __device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } __device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. __device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; - uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { uint32_t gid = g_ * 32u + lane; uint32_t gbase = baseNonce + g_ * 32u; @@ -62,7 +62,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + { uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl r4 = r0 * r2 + r4; // 19 mad @@ -78,9 +78,9 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add r6 = rotr_var(r6, r7); // 30 rotr r3 = r3 ^ ds[r1 & mask]; // 31 load - { uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch + r1 = r1 ^ ds[r0 & mask]; // 32 load r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + { uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor r4 = r4 ^ ds[r0 & mask]; // 37 load @@ -105,7 +105,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba r2 = r2 ^ ds[r7 & mask]; // 56 load r5 = r5 - r6; // 57 sub r1 = r1 ^ ds[r3 & mask]; // 58 load - { uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch + r1 = r1 ^ ds[r4 & mask]; // 59 load r4 = r4 - r6; // 60 sub r1 = r1 * r2; // 61 mul r3 = r6 * r0 + r3; // 62 mad diff --git a/proto-cuda/packs-readwidth/scr4/memhard.h b/proto-cuda/packs-readwidth/scr2k32/memhard.h similarity index 100% rename from proto-cuda/packs-readwidth/scr4/memhard.h rename to proto-cuda/packs-readwidth/scr2k32/memhard.h diff --git a/proto-cuda/packs-readwidth/scr4/memhard.metal b/proto-cuda/packs-readwidth/scr2k32/memhard.metal similarity index 100% rename from proto-cuda/packs-readwidth/scr4/memhard.metal rename to proto-cuda/packs-readwidth/scr2k32/memhard.metal diff --git a/proto-cuda/packs-readwidth/scr2/program.h b/proto-cuda/packs-readwidth/scr2k32/program.h similarity index 91% rename from proto-cuda/packs-readwidth/scr2/program.h rename to proto-cuda/packs-readwidth/scr2k32/program.h index 4892b7988..b7f0eb488 100644 --- a/proto-cuda/packs-readwidth/scr2/program.h +++ b/proto-cuda/packs-readwidth/scr2k32/program.h @@ -15,7 +15,7 @@ #define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" #define IGNEUM_GENERATOR 2 #define IGNEUM_PROGRAM_ATTEMPT 0 -#define IGNEUM_PROGRAM_ID 0x2f0988e568f37cc3ull +#define IGNEUM_PROGRAM_ID 0xe0b080d155bd35b9ull #define IGNEUM_DAY_STRING "2026-10-03" #define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" #define IGNEUM_DAY0 0x3067619fu @@ -30,20 +30,20 @@ #define IGNEUM_OP_MIX "load=14 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 scratch=2 rotl=1" // Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads // the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x. -#define IGNEUM_LOAD_CLASS "scr2" +#define IGNEUM_LOAD_CLASS "scr2k32" #define IGNEUM_LOAD_SLOTS 16 #define IGNEUM_LOAD_MIX { 100, 0, 0 } #define IGNEUM_LOAD_WIDTH_COUNTS { 14, 0, 0 } // loads of 4, 16, 64 bytes per program #define IGNEUM_BYTES_PER_HASH 448 #define IGNEUM_FOLD_ROT 11 #define IGNEUM_FOLD_MUL 0x9e3779b1u -// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch, +// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch, // groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit). #define IGNEUM_PERSISTENT_WARPS 1 #define IGNEUM_SCRATCH_OPS 2 // scratch read-modify-writes per program (16 per hash) -#define IGNEUM_SCRATCH_SLOTS 2048u -#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u -#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u +#define IGNEUM_SCRATCH_SLOTS 64u +#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u +#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u // 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) #define IGNEUM_DATASET_MODE 1 diff --git a/proto-cuda/packs-readwidth/scr2k32/program.json b/proto-cuda/packs-readwidth/scr2k32/program.json new file mode 100644 index 000000000..e9e8d9cbf --- /dev/null +++ b/proto-cuda/packs-readwidth/scr2k32/program.json @@ -0,0 +1,130 @@ +{ + "format": "igneum-program-pack-3", + "generator": 2, + "attempt": 0, + "program_id": "0xe0b080d155bd35b9", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-genesis", + "seed_bytes": "69676e65756d2d67656e65736973", + "seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 128, + "load_class": "scr2k32", + "load_slots": 16, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [14, 0, 0], + "bytes_per_hash": 448, + "scratch_ops_per_hash": 16, + "scratch_kib_per_warp": 32, + "scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)", + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "op_mix": {"load": 14, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "scratch": 2, "rotl": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "2026-10-03", + "day_bytes": "6461792f323032362d31302d3033", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0x3067619f", + "d1": "0x3c269176", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "instructions": [ + {"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1}, + {"i": 1, "op": "add", "dst": 1, "src": 7, "src2": 2, "imm": "0x42da7657", "imm2": "0xc3bd2355", "rot": 25, "bit": 4, "mask": 16, "width": 1}, + {"i": 2, "op": "add", "dst": 2, "src": 3, "src2": 0, "imm": "0x61f0b51c", "imm2": "0x2735a174", "rot": 4, "bit": 26, "mask": 2, "width": 1}, + {"i": 3, "op": "mad", "dst": 4, "src": 0, "src2": 6, "imm": "0x679648a8", "imm2": "0x3044ba32", "rot": 31, "bit": 31, "mask": 4, "width": 1}, + {"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1}, + {"i": 5, "op": "load", "dst": 4, "src": 1, "src2": 2, "imm": "0x987c017a", "imm2": "0xf4d60559", "rot": 2, "bit": 0, "mask": 4, "width": 1}, + {"i": 6, "op": "shfl", "dst": 6, "src": 3, "src2": 7, "imm": "0x6ea7b2df", "imm2": "0x9fce5071", "rot": 7, "bit": 15, "mask": 4, "width": 1}, + {"i": 7, "op": "shfl", "dst": 1, "src": 5, "src2": 1, "imm": "0x26a2ecde", "imm2": "0xfec6ad22", "rot": 15, "bit": 11, "mask": 8, "width": 1}, + {"i": 8, "op": "xor", "dst": 7, "src": 5, "src2": 2, "imm": "0xbe4b445c", "imm2": "0x17a5a9c7", "rot": 8, "bit": 8, "mask": 1, "width": 1}, + {"i": 9, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1}, + {"i": 10, "op": "or", "dst": 1, "src": 2, "src2": 3, "imm": "0x4e7dc10d", "imm2": "0x196d165c", "rot": 14, "bit": 27, "mask": 16, "width": 1}, + {"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 1, "imm": "0xc5c3b55d", "imm2": "0xec061424", "rot": 26, "bit": 27, "mask": 8, "width": 1}, + {"i": 12, "op": "or", "dst": 6, "src": 2, "src2": 3, "imm": "0x306542fe", "imm2": "0x1bb1b429", "rot": 31, "bit": 0, "mask": 2, "width": 1}, + {"i": 13, "op": "mul", "dst": 2, "src": 5, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1}, + {"i": 14, "op": "load", "dst": 1, "src": 2, "src2": 5, "imm": "0x028b4d37", "imm2": "0x7bbd78ea", "rot": 15, "bit": 2, "mask": 8, "width": 1}, + {"i": 15, "op": "rotl", "dst": 7, "src": 6, "src2": 6, "imm": "0x5c88a1a7", "imm2": "0x5c628769", "rot": 1, "bit": 3, "mask": 8, "width": 1}, + {"i": 16, "op": "scratch", "dst": 3, "src": 6, "src2": 7, "imm": "0xbac2ae81", "imm2": "0xcbbc7bdb", "rot": 18, "bit": 8, "mask": 8, "width": 1}, + {"i": 17, "op": "load", "dst": 7, "src": 4, "src2": 2, "imm": "0xe8ab93e9", "imm2": "0xa00de107", "rot": 2, "bit": 1, "mask": 16, "width": 1}, + {"i": 18, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1}, + {"i": 19, "op": "mad", "dst": 4, "src": 0, "src2": 2, "imm": "0x5fba7bc2", "imm2": "0xdf099cfb", "rot": 4, "bit": 15, "mask": 16, "width": 1}, + {"i": 20, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0x0a3056de", "imm2": "0x7f0c25c3", "rot": 27, "bit": 13, "mask": 8, "width": 1}, + {"i": 21, "op": "xor", "dst": 5, "src": 7, "src2": 4, "imm": "0xbd066e1d", "imm2": "0x6d3ddc5a", "rot": 2, "bit": 29, "mask": 1, "width": 1}, + {"i": 22, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1}, + {"i": 23, "op": "load", "dst": 3, "src": 7, "src2": 2, "imm": "0xc7fcfc8f", "imm2": "0x8528b94f", "rot": 17, "bit": 13, "mask": 4, "width": 1}, + {"i": 24, "op": "mulhi", "dst": 7, "src": 3, "src2": 5, "imm": "0xd91641e8", "imm2": "0xaf77faf2", "rot": 22, "bit": 21, "mask": 1, "width": 1}, + {"i": 25, "op": "or", "dst": 5, "src": 4, "src2": 0, "imm": "0x84c03868", "imm2": "0xf6c691b7", "rot": 29, "bit": 14, "mask": 8, "width": 1}, + {"i": 26, "op": "mad", "dst": 4, "src": 5, "src2": 2, "imm": "0x3bb2b6ba", "imm2": "0x49d95fd5", "rot": 1, "bit": 5, "mask": 8, "width": 1}, + {"i": 27, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1}, + {"i": 28, "op": "mulhi", "dst": 6, "src": 7, "src2": 6, "imm": "0xd69c4715", "imm2": "0xe0ebc4ce", "rot": 29, "bit": 2, "mask": 8, "width": 1}, + {"i": 29, "op": "add", "dst": 6, "src": 1, "src2": 2, "imm": "0x3b2d2124", "imm2": "0x187a9128", "rot": 1, "bit": 9, "mask": 16, "width": 1}, + {"i": 30, "op": "rotr", "dst": 6, "src": 7, "src2": 0, "imm": "0x5c64a589", "imm2": "0x61c9a38d", "rot": 17, "bit": 21, "mask": 16, "width": 1}, + {"i": 31, "op": "load", "dst": 3, "src": 1, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1}, + {"i": 32, "op": "load", "dst": 1, "src": 0, "src2": 7, "imm": "0xcc7972c4", "imm2": "0xad098d15", "rot": 30, "bit": 21, "mask": 8, "width": 1}, + {"i": 33, "op": "add", "dst": 0, "src": 4, "src2": 4, "imm": "0x2c35699f", "imm2": "0x351dde38", "rot": 21, "bit": 18, "mask": 4, "width": 1}, + {"i": 34, "op": "scratch", "dst": 0, "src": 2, "src2": 3, "imm": "0xfae8902b", "imm2": "0x5cd8306f", "rot": 5, "bit": 28, "mask": 16, "width": 1}, + {"i": 35, "op": "mul", "dst": 0, "src": 3, "src2": 1, "imm": "0x4fa3f3db", "imm2": "0xdbf37e75", "rot": 7, "bit": 18, "mask": 4, "width": 1}, + {"i": 36, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1}, + {"i": 37, "op": "load", "dst": 4, "src": 0, "src2": 0, "imm": "0x04cc1d55", "imm2": "0x35c52d04", "rot": 11, "bit": 14, "mask": 2, "width": 1}, + {"i": 38, "op": "mad", "dst": 1, "src": 3, "src2": 5, "imm": "0x3958f280", "imm2": "0x8713c7e1", "rot": 5, "bit": 23, "mask": 16, "width": 1}, + {"i": 39, "op": "add", "dst": 0, "src": 3, "src2": 3, "imm": "0xa907b90b", "imm2": "0x1b053acf", "rot": 30, "bit": 25, "mask": 16, "width": 1}, + {"i": 40, "op": "rotr", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1}, + {"i": 41, "op": "mul", "dst": 3, "src": 2, "src2": 4, "imm": "0x49087d74", "imm2": "0x6348b489", "rot": 17, "bit": 9, "mask": 16, "width": 1}, + {"i": 42, "op": "add", "dst": 1, "src": 5, "src2": 1, "imm": "0xa32e000c", "imm2": "0x6058c2e3", "rot": 25, "bit": 20, "mask": 8, "width": 1}, + {"i": 43, "op": "xor", "dst": 3, "src": 4, "src2": 2, "imm": "0x3dad0eb6", "imm2": "0xb97578cb", "rot": 3, "bit": 27, "mask": 1, "width": 1}, + {"i": 44, "op": "load", "dst": 3, "src": 5, "src2": 7, "imm": "0x374aec92", "imm2": "0x626f11df", "rot": 20, "bit": 18, "mask": 8, "width": 1}, + {"i": 45, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1}, + {"i": 46, "op": "xor", "dst": 7, "src": 1, "src2": 0, "imm": "0xef6ac348", "imm2": "0x963bb7e6", "rot": 26, "bit": 3, "mask": 8, "width": 1}, + {"i": 47, "op": "add", "dst": 0, "src": 3, "src2": 0, "imm": "0x838b5065", "imm2": "0x36360066", "rot": 3, "bit": 31, "mask": 4, "width": 1}, + {"i": 48, "op": "mulhi", "dst": 7, "src": 5, "src2": 0, "imm": "0x8458f7ac", "imm2": "0xc1c15026", "rot": 27, "bit": 15, "mask": 8, "width": 1}, + {"i": 49, "op": "load", "dst": 0, "src": 2, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1}, + {"i": 50, "op": "sub", "dst": 2, "src": 6, "src2": 0, "imm": "0x2baec8c9", "imm2": "0x4390f156", "rot": 3, "bit": 12, "mask": 8, "width": 1}, + {"i": 51, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x19234061", "imm2": "0xe84dfade", "rot": 4, "bit": 19, "mask": 1, "width": 1}, + {"i": 52, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0xdc2cd71e", "imm2": "0x1b5d334b", "rot": 9, "bit": 8, "mask": 8, "width": 1}, + {"i": 53, "op": "sub", "dst": 7, "src": 0, "src2": 4, "imm": "0x605c31ec", "imm2": "0x9923ff88", "rot": 28, "bit": 25, "mask": 4, "width": 1}, + {"i": 54, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1}, + {"i": 55, "op": "xor", "dst": 7, "src": 5, "src2": 5, "imm": "0xad7493e7", "imm2": "0x3e400372", "rot": 13, "bit": 8, "mask": 1, "width": 1}, + {"i": 56, "op": "load", "dst": 2, "src": 7, "src2": 1, "imm": "0x87e933c9", "imm2": "0x8c854c1b", "rot": 17, "bit": 3, "mask": 8, "width": 1}, + {"i": 57, "op": "sub", "dst": 5, "src": 6, "src2": 5, "imm": "0x11be3bc9", "imm2": "0xbbaa8e24", "rot": 6, "bit": 5, "mask": 16, "width": 1}, + {"i": 58, "op": "load", "dst": 1, "src": 3, "src2": 2, "imm": "0xa732351a", "imm2": "0xc01349cd", "rot": 14, "bit": 17, "mask": 16, "width": 1}, + {"i": 59, "op": "load", "dst": 1, "src": 4, "src2": 0, "imm": "0xb20547b2", "imm2": "0xc94655de", "rot": 27, "bit": 30, "mask": 1, "width": 1}, + {"i": 60, "op": "sub", "dst": 4, "src": 6, "src2": 7, "imm": "0x67cf904c", "imm2": "0x6873b216", "rot": 27, "bit": 7, "mask": 16, "width": 1}, + {"i": 61, "op": "mul", "dst": 1, "src": 2, "src2": 7, "imm": "0x93ab0bf4", "imm2": "0x96158375", "rot": 14, "bit": 0, "mask": 16, "width": 1}, + {"i": 62, "op": "mad", "dst": 3, "src": 6, "src2": 0, "imm": "0x41a443a3", "imm2": "0xe69d7919", "rot": 9, "bit": 0, "mask": 16, "width": 1}, + {"i": 63, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x2fe0e98b", "imm2": "0xc88e2942", "rot": 5, "bit": 16, "mask": 16, "width": 1} + ] +} diff --git a/proto-cuda/packs-readwidth/scr4/program.metal b/proto-cuda/packs-readwidth/scr2k32/program.metal similarity index 72% rename from proto-cuda/packs-readwidth/scr4/program.metal rename to proto-cuda/packs-readwidth/scr2k32/program.metal index 4a8922643..453ad7237 100644 --- a/proto-cuda/packs-readwidth/scr4/program.metal +++ b/proto-cuda/packs-readwidth/scr2k32/program.metal @@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) { return x; } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], uint lane = tid & 31u; uint warp_ = tid >> 5; uint nwarps_ = nthreads >> 5; - device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -70,7 +70,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], r2 = r2 * r5; // 13 r1 = r1 ^ dataset[r2 & MASK]; // 14 r7 = rotl_imm(r7, 1u); // 15 - { uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + { uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 r7 = r7 ^ dataset[r4 & MASK]; // 17 r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 r4 = r0 * r2 + r4; // 19 @@ -86,9 +86,9 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29 r6 = rotr_var(r6, r7); // 30 r3 = r3 ^ dataset[r1 & MASK]; // 31 - { uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 + r1 = r1 ^ dataset[r0 & MASK]; // 32 r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 - { uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + { uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 r0 = r0 * r3; // 35 r2 = r2 ^ r5; // 36 r4 = r4 ^ dataset[r0 & MASK]; // 37 @@ -113,7 +113,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], r2 = r2 ^ dataset[r7 & MASK]; // 56 r5 = r5 - r6; // 57 r1 = r1 ^ dataset[r3 & MASK]; // 58 - { uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 + r1 = r1 ^ dataset[r4 & MASK]; // 59 r4 = r4 - r6; // 60 r1 = r1 * r2; // 61 r3 = r6 * r0 + r3; // 62 diff --git a/proto-cuda/packs-readwidth/scr4/program_bound.metal b/proto-cuda/packs-readwidth/scr2k32/program_bound.metal similarity index 73% rename from proto-cuda/packs-readwidth/scr4/program_bound.metal rename to proto-cuda/packs-readwidth/scr2k32/program_bound.metal index 6235bce84..24e7f8b4d 100644 --- a/proto-cuda/packs-readwidth/scr4/program_bound.metal +++ b/proto-cuda/packs-readwidth/scr2k32/program_bound.metal @@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) { return x; } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], uint lane = tid & 31u; uint warp_ = tid >> 5; uint nwarps_ = nthreads >> 5; - device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -72,7 +72,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], r2 = r2 * r5; // 13 r1 = r1 ^ dataset[r2 & MASK]; // 14 r7 = rotl_imm(r7, 1u); // 15 - { uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + { uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 r7 = r7 ^ dataset[r4 & MASK]; // 17 r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 r4 = r0 * r2 + r4; // 19 @@ -88,9 +88,9 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29 r6 = rotr_var(r6, r7); // 30 r3 = r3 ^ dataset[r1 & MASK]; // 31 - { uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 + r1 = r1 ^ dataset[r0 & MASK]; // 32 r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 - { uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + { uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 r0 = r0 * r3; // 35 r2 = r2 ^ r5; // 36 r4 = r4 ^ dataset[r0 & MASK]; // 37 @@ -115,7 +115,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], r2 = r2 ^ dataset[r7 & MASK]; // 56 r5 = r5 - r6; // 57 r1 = r1 ^ dataset[r3 & MASK]; // 58 - { uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 + r1 = r1 ^ dataset[r4 & MASK]; // 59 r4 = r4 - r6; // 60 r1 = r1 * r2; // 61 r3 = r6 * r0 + r3; // 62 diff --git a/proto-cuda/packs-readwidth/scr8/vectors.h b/proto-cuda/packs-readwidth/scr2k32/vectors.h similarity index 60% rename from proto-cuda/packs-readwidth/scr8/vectors.h rename to proto-cuda/packs-readwidth/scr2k32/vectors.h index 43d22af98..e65ab66b4 100644 --- a/proto-cuda/packs-readwidth/scr8/vectors.h +++ b/proto-cuda/packs-readwidth/scr2k32/vectors.h @@ -11,22 +11,22 @@ static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { { // base nonce 0 - 0xd0f846c3cd57ae09ull, 0x4d5e1caf41761a5cull, 0x7fc77bb7fa221a51ull, 0xe45b6317d4f52ae7ull, 0x09704a39107a3150ull, 0xd5166620f9dc49abull, 0xeaf1fa69e4075e55ull, 0x38e23d449b7616aeull, - 0xf4593424c8427320ull, 0x9359a44a5149bd60ull, 0x2df93b5bc274f083ull, 0xfd80eb32016f8659ull, 0xdeb10b8a37fc3ee7ull, 0xc18e2ed77ab77f34ull, 0x7f39b618cc81c1b4ull, 0xdc1b7299961ad5abull, - 0x8f9c03d151013ec4ull, 0x668927a4a267d76full, 0x233a50c329caf635ull, 0x7c80406541ae3f15ull, 0x18fb38606c48af4aull, 0x0452a913df5f112bull, 0x59065cb670ba6d7dull, 0xa2998e2ccd726df8ull, - 0xef690e221af37893ull, 0x4e87968e5c45903eull, 0x6c6e5d4c1b35d7a9ull, 0x54d7e322426dcae3ull, 0x0d3ed6c08deda320ull, 0x3d6a4301fbb06a5aull, 0x9eb78b04d2366566ull, 0x7806f8c64d2d09ffull + 0x589f62cd61c27dc1ull, 0xf6be7bab8b00a7c7ull, 0x349551c5979e330eull, 0x6d8156f5afaf0064ull, 0x81971552315d55b8ull, 0x20a3bb31ef5e202cull, 0x91818c4ec9fd5fecull, 0x35ca8cde74ac9715ull, + 0xc1db0a90f80c8801ull, 0xbc52521a33053a8aull, 0x89fda590bf946dd6ull, 0xc54fe4aeaf11975full, 0xffc712960d7f4022ull, 0x4da6b3de6f9a0abdull, 0xf70ff0468e9b595aull, 0xc2d1d4434c1eb8c9ull, + 0x70dc278b36920857ull, 0xfb20a2d85fa65b83ull, 0xd24316c07937542dull, 0xd505e1e3e85694bdull, 0x88c759248ae122a7ull, 0xf1c04d8ad71d5e8cull, 0x7e4d0e4e99717f12ull, 0x63d9e4b8f619ac98ull, + 0xcfef2a4243ba8608ull, 0xb2908c3da3593ba9ull, 0x48bee70b2ee45b84ull, 0x8f7a09d71be2c321ull, 0x0aadd48107e05bdbull, 0x8b73ff0f34ddaaaaull, 0x9ff3a4875ecf0fe3ull, 0xaf3d701cd1cbbd4bull }, { // base nonce 4096 - 0xe2c97c384a85c687ull, 0xcf905b005ab01ebcull, 0x4526799fae210c6bull, 0xd4daf72ed75e8a16ull, 0xb3703a9e7c6820a8ull, 0xb6be9393fbb920bdull, 0xe2fff04fb7816eb8ull, 0xed76ad5dbf400f91ull, - 0xea7d4b58ab0eb8d1ull, 0xf67a73b01f030532ull, 0x35ab823037510099ull, 0x5de1b1b0b3a26c69ull, 0xbe5dd90dbb7e632bull, 0x1475918a9237e24dull, 0x177fd53c45634d71ull, 0xa7cf00759ba28ce0ull, - 0xe51be0584ac3fbb4ull, 0x427049cc778aab35ull, 0x826bab125577d172ull, 0xd705b891b16237f5ull, 0xdf622fd44b180a87ull, 0x359398ecb79ec2deull, 0x2a1e075fb078da66ull, 0xd480ddd8e66d26e8ull, - 0x07e3e86f517de466ull, 0xa98b2f7423557445ull, 0x6bab15b36bb142feull, 0xf87d147bf2cc5c0bull, 0xf0294ea2b2820e03ull, 0xf219ac95e823d794ull, 0x9fa3fea85bc54264ull, 0xc4af3fafd2ff5201ull + 0xc93652d639480287ull, 0x2bff33a7119c0ec9ull, 0x7c676a1cf9f37474ull, 0xcb1dab3b216bdd52ull, 0x236552ac169e0f4aull, 0xffb7303a10cb2833ull, 0xdf1ca9f83cf0740cull, 0xe88fcc23bd9a0a9full, + 0x48118af81df77459ull, 0x49379fea0c36ec78ull, 0x931e8c0ca930cd35ull, 0xba3e0b2487710abfull, 0x56b607c6f3672398ull, 0xeb0215d7735482c3ull, 0x104b2405ae428d28ull, 0xa8c21e4eb7e2b744ull, + 0xbc41f10ffa4e4621ull, 0xc51ab216e6e0a339ull, 0x285cdde4bd22e712ull, 0x4f985d36d1302ebdull, 0x82563e5cd9a31b86ull, 0x9495c481dd399662ull, 0xec6111e88d79f207ull, 0x112bf6a954166121ull, + 0x1a6eda3d1a3846b6ull, 0xd25ebd17cbfaf07dull, 0xe552179181d4390full, 0xdd0129cf2d8db153ull, 0x0863f60becfbabedull, 0x1825c21e13698cecull, 0x20acd589e0408f6eull, 0x817a8413e24a68d4ull }, { // base nonce 1000000 - 0x501f772483fac0a3ull, 0x461363da2c1539e0ull, 0x750050674de592afull, 0x14ed105042cd912cull, 0xc6477310878614ebull, 0xe916b44e32e90e56ull, 0x531c92e69c2bdd73ull, 0x5bb0129ef7c3bd51ull, - 0x967ed91f7cbc7c79ull, 0x06fcc6a895b58b4aull, 0x3bdb29fbf93cbff3ull, 0xbed385172d2e6abeull, 0x921fc99ff4efac5eull, 0x6b0090bedd9f69c7ull, 0x0b105c18aaa53ab0ull, 0xf0c4203561f3b94dull, - 0x64abe33adccf7807ull, 0xd8f3a3b7e0242c04ull, 0x397da462f69fac5bull, 0xe6b6b32e467b71beull, 0x5318ac9e56278d04ull, 0xf7c5e348a4e1f5dbull, 0x79c994e6109646dfull, 0x9cdc42b0f6e82231ull, - 0x3f1160f02fd96ac7ull, 0xa69a23fc7058be21ull, 0xccde0a19bfdc25b6ull, 0xd3684a5b966f7497ull, 0x929db00b97a624f9ull, 0xfd14a40882d395a6ull, 0x49b0c1ecc514d6adull, 0x95d994c8349be12dull + 0x7535b29ea3e2823eull, 0xd84fe0e0281b6538ull, 0x14f4b14bf5185a26ull, 0x5fc4ec481d8c65bcull, 0x9ff6a4ec626c4cbcull, 0x52131ecd506117f0ull, 0x9a7db1822213f9b9ull, 0x025ac827f2f88c7cull, + 0x4076da8ca02131e1ull, 0xbc9956bc70d53e0bull, 0x077cf7d297357750ull, 0xb00c8db428fbccf2ull, 0x50f16eecd1fb65c4ull, 0x4daddb3cf455583dull, 0x952b3cca95e87c92ull, 0xa7b7af6eac1a0222ull, + 0xc58c0db5a99ced05ull, 0xa71a8ce697d65e94ull, 0xe29bab54459076d2ull, 0x5f613619a76c6400ull, 0xb43e9559e242a8d4ull, 0x4b5433e68aa1f302ull, 0x382b7f105840032cull, 0xbf402649fb8e9618ull, + 0xf45994bb29726a41ull, 0x8a11f358c24795cbull, 0xc2c4f8902007527cull, 0xe12f65396a832dcdull, 0x307d3f495790aff0ull, 0x5fc00c5eb0c3e81eull, 0xb46600e4685191eeull, 0xce65db0e1d36f875ull } }; diff --git a/proto-cuda/packs-readwidth/scr2/vectors.json b/proto-cuda/packs-readwidth/scr2k32/vectors.json similarity index 65% rename from proto-cuda/packs-readwidth/scr2/vectors.json rename to proto-cuda/packs-readwidth/scr2k32/vectors.json index d30ae7d93..c32644571 100644 --- a/proto-cuda/packs-readwidth/scr2/vectors.json +++ b/proto-cuda/packs-readwidth/scr2k32/vectors.json @@ -8,22 +8,22 @@ "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", "warps": [ {"base_nonce": 0, "expected": [ - "0x2273e2732203e32a", "0xa513d354bd107990", "0xe005b7515054c85f", "0x18a61b37b30cd1fb", "0xa21d5b98e8d07e9c", "0x5c24171a391d5ed0", "0x0f29e7583e1794b8", "0x7eca8a374d1a4f70", - "0x260ca011cd9ea10c", "0xce1050798fce3d43", "0x9560d939dca19041", "0x8480a8440b80ecc3", "0xfaa99aac459b739e", "0x7f083e72458e08ab", "0x78d876842f68672b", "0x3b9bcf6275d3575c", - "0x0256af61bdbf11b3", "0xefc6771cae646cbd", "0xbc44f1c9f9f54d87", "0x6caedd783487eb7d", "0x001b31fcb4fe0d4d", "0x947a7ba1057e25b6", "0xb9e5a0204d68c22a", "0x50489bed25d42661", - "0xb1019bff6d1057cd", "0xd1442990562ce940", "0xcd986a47f98801db", "0x9c8796b6df23300f", "0xbd53ad05d2c877a9", "0xc95e863774a15b0a", "0x132d8a91fb2fa67a", "0x53dd38e8eadc8a24" + "0x589f62cd61c27dc1", "0xf6be7bab8b00a7c7", "0x349551c5979e330e", "0x6d8156f5afaf0064", "0x81971552315d55b8", "0x20a3bb31ef5e202c", "0x91818c4ec9fd5fec", "0x35ca8cde74ac9715", + "0xc1db0a90f80c8801", "0xbc52521a33053a8a", "0x89fda590bf946dd6", "0xc54fe4aeaf11975f", "0xffc712960d7f4022", "0x4da6b3de6f9a0abd", "0xf70ff0468e9b595a", "0xc2d1d4434c1eb8c9", + "0x70dc278b36920857", "0xfb20a2d85fa65b83", "0xd24316c07937542d", "0xd505e1e3e85694bd", "0x88c759248ae122a7", "0xf1c04d8ad71d5e8c", "0x7e4d0e4e99717f12", "0x63d9e4b8f619ac98", + "0xcfef2a4243ba8608", "0xb2908c3da3593ba9", "0x48bee70b2ee45b84", "0x8f7a09d71be2c321", "0x0aadd48107e05bdb", "0x8b73ff0f34ddaaaa", "0x9ff3a4875ecf0fe3", "0xaf3d701cd1cbbd4b" ]}, {"base_nonce": 4096, "expected": [ - "0x78c93312a03fb0ee", "0x184fea638ec9b5fb", "0x5687e8dcc4301dbf", "0xed02c94f23681dfc", "0x326d70162241ff6d", "0x452017eb4ed2dfcf", "0xc10b0e016f1e28c9", "0x691ce0cecf2a99ba", - "0x6c9506f34e0e63ce", "0x447a98c2b7fdfa40", "0x07486b0e4055b2c9", "0x41781460bd47fd5c", "0x01db316e35198291", "0xccd7e727f139a880", "0xdd7bd9efd16bf21c", "0x8285d37966656366", - "0x383deade15fe0ecb", "0x5fd64f5873c8e324", "0xad584cb6839c5e1d", "0xbb842707fb5e9460", "0x4e8bc8f87978fcbd", "0x18eb56f4a1fae881", "0x4c3b731a6b0c47a1", "0xda52cf9d69b252eb", - "0xb5ff19b2b3eeb13e", "0xe2595cea2afe42dd", "0x3ff108424c9e6e38", "0x3a8a9e1995f359ca", "0x6a6b1da662cf2126", "0x54e684c127bb181f", "0x2018caa81f1a7d50", "0x33e94d2c92d9d148" + "0xc93652d639480287", "0x2bff33a7119c0ec9", "0x7c676a1cf9f37474", "0xcb1dab3b216bdd52", "0x236552ac169e0f4a", "0xffb7303a10cb2833", "0xdf1ca9f83cf0740c", "0xe88fcc23bd9a0a9f", + "0x48118af81df77459", "0x49379fea0c36ec78", "0x931e8c0ca930cd35", "0xba3e0b2487710abf", "0x56b607c6f3672398", "0xeb0215d7735482c3", "0x104b2405ae428d28", "0xa8c21e4eb7e2b744", + "0xbc41f10ffa4e4621", "0xc51ab216e6e0a339", "0x285cdde4bd22e712", "0x4f985d36d1302ebd", "0x82563e5cd9a31b86", "0x9495c481dd399662", "0xec6111e88d79f207", "0x112bf6a954166121", + "0x1a6eda3d1a3846b6", "0xd25ebd17cbfaf07d", "0xe552179181d4390f", "0xdd0129cf2d8db153", "0x0863f60becfbabed", "0x1825c21e13698cec", "0x20acd589e0408f6e", "0x817a8413e24a68d4" ]}, {"base_nonce": 1000000, "expected": [ - "0x04a41389bf3dfd3d", "0xc509164def9207df", "0x4a8ffdbdf46e429d", "0xff13bf0dc1b39aeb", "0xb852acc8e24133d7", "0x4bdd991ae56252ac", "0xa7739e74b3a054e9", "0xb4e36218d4b45fdc", - "0x8bbd323155f5edc5", "0xb7b56a90659e7fd2", "0xdff7c495b7027480", "0xffa8adb5c0302b06", "0xe97d7967d89a5672", "0x0d0c2d4e6493926e", "0xe9a5cda333cf2043", "0xdc95256d0986e5d8", - "0xd0dc211b811d6843", "0x68dfa3d0fb9a569b", "0xa9e0028dfd9178c0", "0x4a36ca1fc40b20a9", "0xe7c765c5a735294b", "0xf08954b015cb2628", "0xc69ee66ecf2740c5", "0xe3d01e899e46b089", - "0xc3558c74159c8603", "0x4c7aeb196bd01b04", "0x13c17119385f1910", "0xda7fca98e0989b8a", "0x6f95baf340817945", "0x1af52756fd3afcab", "0xb8eefc370bbe7e4b", "0x94a55055b48bd4db" + "0x7535b29ea3e2823e", "0xd84fe0e0281b6538", "0x14f4b14bf5185a26", "0x5fc4ec481d8c65bc", "0x9ff6a4ec626c4cbc", "0x52131ecd506117f0", "0x9a7db1822213f9b9", "0x025ac827f2f88c7c", + "0x4076da8ca02131e1", "0xbc9956bc70d53e0b", "0x077cf7d297357750", "0xb00c8db428fbccf2", "0x50f16eecd1fb65c4", "0x4daddb3cf455583d", "0x952b3cca95e87c92", "0xa7b7af6eac1a0222", + "0xc58c0db5a99ced05", "0xa71a8ce697d65e94", "0xe29bab54459076d2", "0x5f613619a76c6400", "0xb43e9559e242a8d4", "0x4b5433e68aa1f302", "0x382b7f105840032c", "0xbf402649fb8e9618", + "0xf45994bb29726a41", "0x8a11f358c24795cb", "0xc2c4f8902007527c", "0xe12f65396a832dcd", "0x307d3f495790aff0", "0x5fc00c5eb0c3e81e", "0xb46600e4685191ee", "0xce65db0e1d36f875" ]} ], "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], diff --git a/proto-cuda/packs-readwidth/scr8/kernel.cl b/proto-cuda/packs-readwidth/scr4k128/kernel.cl similarity index 78% rename from proto-cuda/packs-readwidth/scr8/kernel.cl rename to proto-cuda/packs-readwidth/scr4k128/kernel.cl index 50015ca9b..3917a927d 100644 --- a/proto-cuda/packs-readwidth/scr8/kernel.cl +++ b/proto-cuda/packs-readwidth/scr4k128/kernel.cl @@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n // One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the // lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and // __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -214,7 +214,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add r4 = r0 * r6 + r4; // 3 mad - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch + r7 = r7 ^ ds[r2 & mask]; // 4 load r4 = r4 ^ ds[r1 & mask]; // 5 load { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl @@ -226,14 +226,14 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + { uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl r4 = r0 * r2 + r4; // 19 mad { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl r5 = r5 ^ r7; // 21 xor r2 = mul_hi(r2, r5); // 22 mulhi - { uint s_ = r7 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch + r3 = r3 ^ ds[r7 & mask]; // 23 load r7 = mul_hi(r7, r3); // 24 mulhi r5 = r5 | r4; // 25 or r4 = r5 * r2 + r4; // 26 mad @@ -242,19 +242,19 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add r6 = rotr_var(r6, r7); // 30 rotr r3 = r3 ^ ds[r1 & mask]; // 31 load - { uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + { uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor - { uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch + r4 = r4 ^ ds[r0 & mask]; // 37 load r1 = r3 * r5 + r1; // 38 mad r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add r2 = rotr_var(r2, r5); // 40 rotr r3 = r3 * r2; // 41 mul r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add r3 = r3 ^ r4; // 43 xor - { uint s_ = r5 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch + r3 = r3 ^ ds[r5 & mask]; // 44 load r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add r7 = r7 ^ r1; // 46 xor r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add @@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r2 = r2 ^ ds[r7 & mask]; // 56 load r5 = r5 - r6; // 57 sub r1 = r1 ^ ds[r3 & mask]; // 58 load - { uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + { uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch r4 = r4 - r6; // 60 sub r1 = r1 * r2; // 61 mul r3 = r6 * r0 + r3; // 62 mad diff --git a/proto-cuda/packs-readwidth/scr8/kernel.cu b/proto-cuda/packs-readwidth/scr4k128/kernel.cu similarity index 66% rename from proto-cuda/packs-readwidth/scr8/kernel.cu rename to proto-cuda/packs-readwidth/scr4k128/kernel.cu index 85212204b..67b3ff7d8 100644 --- a/proto-cuda/packs-readwidth/scr8/kernel.cu +++ b/proto-cuda/packs-readwidth/scr4k128/kernel.cu @@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem // One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every // __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a // 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. __device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; - uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { uint32_t gid = g_ * 32u + lane; uint32_t gbase = baseNonce + g_ * 32u; @@ -74,7 +74,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add r4 = r0 * r6 + r4; // 3 mad - { uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch + r7 = r7 ^ ds[r2 & mask]; // 4 load r4 = r4 ^ ds[r1 & mask]; // 5 load r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl @@ -86,14 +86,14 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + { uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl r4 = r0 * r2 + r4; // 19 mad r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl r5 = r5 ^ r7; // 21 xor r2 = __umulhi(r2, r5); // 22 mulhi - { uint32_t s_ = r7 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch + r3 = r3 ^ ds[r7 & mask]; // 23 load r7 = __umulhi(r7, r3); // 24 mulhi r5 = r5 | r4; // 25 or r4 = r5 * r2 + r4; // 26 mad @@ -102,19 +102,19 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add r6 = rotr_var(r6, r7); // 30 rotr r3 = r3 ^ ds[r1 & mask]; // 31 load - { uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch + { uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + { uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor - { uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch + r4 = r4 ^ ds[r0 & mask]; // 37 load r1 = r3 * r5 + r1; // 38 mad r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add r2 = rotr_var(r2, r5); // 40 rotr r3 = r3 * r2; // 41 mul r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add r3 = r3 ^ r4; // 43 xor - { uint32_t s_ = r5 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch + r3 = r3 ^ ds[r5 & mask]; // 44 load r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add r7 = r7 ^ r1; // 46 xor r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add @@ -129,7 +129,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc r2 = r2 ^ ds[r7 & mask]; // 56 load r5 = r5 - r6; // 57 sub r1 = r1 ^ ds[r3 & mask]; // 58 load - { uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch + { uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch r4 = r4 - r6; // 60 sub r1 = r1 * r2; // 61 mul r3 = r6 * r0 + r3; // 62 mad diff --git a/proto-cuda/packs-readwidth/scr8/kernel_bound.cl b/proto-cuda/packs-readwidth/scr4k128/kernel_bound.cl similarity index 70% rename from proto-cuda/packs-readwidth/scr8/kernel_bound.cl rename to proto-cuda/packs-readwidth/scr4k128/kernel_bound.cl index 73b2b9c9f..669eb1d51 100644 --- a/proto-cuda/packs-readwidth/scr8/kernel_bound.cl +++ b/proto-cuda/packs-readwidth/scr4k128/kernel_bound.cl @@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n // One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the // lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and // __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -214,7 +214,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add r4 = r0 * r6 + r4; // 3 mad - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch + r7 = r7 ^ ds[r2 & mask]; // 4 load r4 = r4 ^ ds[r1 & mask]; // 5 load { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl @@ -226,14 +226,14 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + { uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl r4 = r0 * r2 + r4; // 19 mad { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl r5 = r5 ^ r7; // 21 xor r2 = mul_hi(r2, r5); // 22 mulhi - { uint s_ = r7 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch + r3 = r3 ^ ds[r7 & mask]; // 23 load r7 = mul_hi(r7, r3); // 24 mulhi r5 = r5 | r4; // 25 or r4 = r5 * r2 + r4; // 26 mad @@ -242,19 +242,19 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add r6 = rotr_var(r6, r7); // 30 rotr r3 = r3 ^ ds[r1 & mask]; // 31 load - { uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + { uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor - { uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch + r4 = r4 ^ ds[r0 & mask]; // 37 load r1 = r3 * r5 + r1; // 38 mad r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add r2 = rotr_var(r2, r5); // 40 rotr r3 = r3 * r2; // 41 mul r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add r3 = r3 ^ r4; // 43 xor - { uint s_ = r5 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch + r3 = r3 ^ ds[r5 & mask]; // 44 load r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add r7 = r7 ^ r1; // 46 xor r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add @@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out r2 = r2 ^ ds[r7 & mask]; // 56 load r5 = r5 - r6; // 57 sub r1 = r1 ^ ds[r3 & mask]; // 58 load - { uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + { uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch r4 = r4 - r6; // 60 sub r1 = r1 * r2; // 61 mul r3 = r6 * r0 + r3; // 62 mad @@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon uint lane = (uint)get_global_id(0) & 31u; uint warp_ = (uint)get_global_id(0) >> 5; uint nwarps_ = (uint)get_global_size(0) >> 5; - __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -325,7 +325,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add r4 = r0 * r6 + r4; // 3 mad - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch + r7 = r7 ^ ds[r2 & mask]; // 4 load r4 = r4 ^ ds[r1 & mask]; // 5 load { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl @@ -337,14 +337,14 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + { uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl r4 = r0 * r2 + r4; // 19 mad { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl r5 = r5 ^ r7; // 21 xor r2 = mul_hi(r2, r5); // 22 mulhi - { uint s_ = r7 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch + r3 = r3 ^ ds[r7 & mask]; // 23 load r7 = mul_hi(r7, r3); // 24 mulhi r5 = r5 | r4; // 25 or r4 = r5 * r2 + r4; // 26 mad @@ -353,19 +353,19 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add r6 = rotr_var(r6, r7); // 30 rotr r3 = r3 ^ ds[r1 & mask]; // 31 load - { uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + { uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor - { uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch + r4 = r4 ^ ds[r0 & mask]; // 37 load r1 = r3 * r5 + r1; // 38 mad r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add r2 = rotr_var(r2, r5); // 40 rotr r3 = r3 * r2; // 41 mul r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add r3 = r3 ^ r4; // 43 xor - { uint s_ = r5 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch + r3 = r3 ^ ds[r5 & mask]; // 44 load r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add r7 = r7 ^ r1; // 46 xor r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add @@ -380,7 +380,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon r2 = r2 ^ ds[r7 & mask]; // 56 load r5 = r5 - r6; // 57 sub r1 = r1 ^ ds[r3 & mask]; // 58 load - { uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + { uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch r4 = r4 - r6; // 60 sub r1 = r1 * r2; // 61 mul r3 = r6 * r0 + r3; // 62 mad diff --git a/proto-cuda/packs-readwidth/scr8/kernel_bound.cu b/proto-cuda/packs-readwidth/scr4k128/kernel_bound.cu similarity index 60% rename from proto-cuda/packs-readwidth/scr8/kernel_bound.cu rename to proto-cuda/packs-readwidth/scr4k128/kernel_bound.cu index 13d40a9fc..9197cfce9 100644 --- a/proto-cuda/packs-readwidth/scr8/kernel_bound.cu +++ b/proto-cuda/packs-readwidth/scr4k128/kernel_bound.cu @@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) { __device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } __device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. __device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; - uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { uint32_t gid = g_ * 32u + lane; uint32_t gbase = baseNonce + g_ * 32u; @@ -50,7 +50,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add r4 = r0 * r6 + r4; // 3 mad - { uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch + r7 = r7 ^ ds[r2 & mask]; // 4 load r4 = r4 ^ ds[r1 & mask]; // 5 load r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl @@ -62,14 +62,14 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba r2 = r2 * r5; // 13 mul r1 = r1 ^ ds[r2 & mask]; // 14 load r7 = rotl_imm(r7, 1u); // 15 rotl - { uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + { uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch r7 = r7 ^ ds[r4 & mask]; // 17 load r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl r4 = r0 * r2 + r4; // 19 mad r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl r5 = r5 ^ r7; // 21 xor r2 = __umulhi(r2, r5); // 22 mulhi - { uint32_t s_ = r7 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch + r3 = r3 ^ ds[r7 & mask]; // 23 load r7 = __umulhi(r7, r3); // 24 mulhi r5 = r5 | r4; // 25 or r4 = r5 * r2 + r4; // 26 mad @@ -78,19 +78,19 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add r6 = rotr_var(r6, r7); // 30 rotr r3 = r3 ^ ds[r1 & mask]; // 31 load - { uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch + { uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add - { uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + { uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch r0 = r0 * r3; // 35 mul r2 = r2 ^ r5; // 36 xor - { uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch + r4 = r4 ^ ds[r0 & mask]; // 37 load r1 = r3 * r5 + r1; // 38 mad r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add r2 = rotr_var(r2, r5); // 40 rotr r3 = r3 * r2; // 41 mul r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add r3 = r3 ^ r4; // 43 xor - { uint32_t s_ = r5 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch + r3 = r3 ^ ds[r5 & mask]; // 44 load r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add r7 = r7 ^ r1; // 46 xor r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add @@ -105,7 +105,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba r2 = r2 ^ ds[r7 & mask]; // 56 load r5 = r5 - r6; // 57 sub r1 = r1 ^ ds[r3 & mask]; // 58 load - { uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch + { uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch r4 = r4 - r6; // 60 sub r1 = r1 * r2; // 61 mul r3 = r6 * r0 + r3; // 62 mad diff --git a/proto-cuda/packs-readwidth/scr8/memhard.h b/proto-cuda/packs-readwidth/scr4k128/memhard.h similarity index 100% rename from proto-cuda/packs-readwidth/scr8/memhard.h rename to proto-cuda/packs-readwidth/scr4k128/memhard.h diff --git a/proto-cuda/packs-readwidth/scr8/memhard.metal b/proto-cuda/packs-readwidth/scr4k128/memhard.metal similarity index 100% rename from proto-cuda/packs-readwidth/scr8/memhard.metal rename to proto-cuda/packs-readwidth/scr4k128/memhard.metal diff --git a/proto-cuda/packs-readwidth/scr4k128/program.h b/proto-cuda/packs-readwidth/scr4k128/program.h new file mode 100644 index 000000000..ee7cc85d0 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k128/program.h @@ -0,0 +1,67 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Program metadata for host.cu plus the launch wrappers defined in kernel.cu. +// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#ifndef IGNEUM_NO_CUDA +#include +#endif + +#define IGNEUM_SEED_STRING "igneum-genesis" +#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" +#define IGNEUM_GENERATOR 2 +#define IGNEUM_PROGRAM_ATTEMPT 0 +#define IGNEUM_PROGRAM_ID 0xe0c444d155cd78cfull +#define IGNEUM_DAY_STRING "2026-10-03" +#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" +#define IGNEUM_DAY0 0x3067619fu +#define IGNEUM_DAY1 0x3c269176u +#define IGNEUM_DATASET_LOG2 28 +#define IGNEUM_MASK 0x0fffffffu +#define IGNEUM_LANES 32 +#define IGNEUM_ITERATIONS 8 +#define IGNEUM_INSTR_COUNT 64 +#define IGNEUM_LOADS_PER_HASH 128 +#define IGNEUM_WIDE_LOADS_PER_HASH 0 +#define IGNEUM_OP_MIX "load=12 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 scratch=4 shfl=4 rotr=2 rotl=1" +// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads +// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x. +#define IGNEUM_LOAD_CLASS "scr4k128" +#define IGNEUM_LOAD_SLOTS 16 +#define IGNEUM_LOAD_MIX { 100, 0, 0 } +#define IGNEUM_LOAD_WIDTH_COUNTS { 12, 0, 0 } // loads of 4, 16, 64 bytes per program +#define IGNEUM_BYTES_PER_HASH 384 +#define IGNEUM_FOLD_ROT 11 +#define IGNEUM_FOLD_MUL 0x9e3779b1u +// Variant 5: persistent warps, a 128 KiB scratch per launched warp (the host launches N warps and passes scratch, +// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit). +#define IGNEUM_PERSISTENT_WARPS 1 +#define IGNEUM_SCRATCH_OPS 4 // scratch read-modify-writes per program (32 per hash) +#define IGNEUM_SCRATCH_SLOTS 256u +#define IGNEUM_SCRATCH_WORDS_PER_LANE 1024u +#define IGNEUM_SCRATCH_BYTES_PER_WARP 131072u +// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) +#define IGNEUM_DATASET_MODE 1 + +#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u } +#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u } +#define IGNEUM_CACHE_LOG2_WORDS 26 +#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6 +#define IGNEUM_CACHE_SEGMENTS 65536u +#define IGNEUM_ITEM_ROUNDS 8 +#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u } +#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u } +#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u } + +#ifndef IGNEUM_NO_CUDA +// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError(). +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt); +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#endif diff --git a/proto-cuda/packs-readwidth/scr4/program.json b/proto-cuda/packs-readwidth/scr4k128/program.json similarity index 98% rename from proto-cuda/packs-readwidth/scr4/program.json rename to proto-cuda/packs-readwidth/scr4k128/program.json index 61fe86581..4d0d0ad78 100644 --- a/proto-cuda/packs-readwidth/scr4/program.json +++ b/proto-cuda/packs-readwidth/scr4k128/program.json @@ -2,7 +2,7 @@ "format": "igneum-program-pack-3", "generator": 2, "attempt": 0, - "program_id": "0x2f098ee568f386f5", + "program_id": "0xe0c444d155cd78cf", "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", "dataset_mode": "memory-hard", "seed": "igneum-genesis", @@ -15,13 +15,14 @@ "iterations": 8, "instruction_count": 64, "loads_per_hash": 128, - "load_class": "scr4", + "load_class": "scr4k128", "load_slots": 16, "load_mix_percent_4_16_64": [100, 0, 0], "load_width_counts_4_16_64": [12, 0, 0], "bytes_per_hash": 384, "scratch_ops_per_hash": 32, - "scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)", + "scratch_kib_per_warp": 128, + "scratch": "variant 5 (measurement only): persistent warps; a 128 KiB scratch per warp of 256 16-byte slots per lane (lane-major); slot = src & 0xff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)", "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", "op_mix": {"load": 12, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "scratch": 4, "shfl": 4, "rotr": 2, "rotl": 1}, "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", diff --git a/proto-cuda/packs-readwidth/scr4k128/program.metal b/proto-cuda/packs-readwidth/scr4k128/program.metal new file mode 100644 index 000000000..f4f56728c --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k128/program.metal @@ -0,0 +1,126 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +kernel void igneum_hash(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + device uint* scratch [[buffer(3)]], + constant uint& groups [[buffer(4)]], + constant uint& salt [[buffer(5)]], + uint tid [[thread_position_in_grid]], + uint nthreads [[threads_per_grid]]) { + uint lane = tid & 31u; + uint warp_ = tid >> 5; + uint nwarps_ = nthreads >> 5; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; } + { uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; } + { uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; } + { uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; } + { uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; } + { uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; } + { uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; } + { uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1 + r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2 + r4 = r0 * r6 + r4; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r4 = r4 ^ dataset[r1 & MASK]; // 5 + r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7 + r7 = r7 ^ r5; // 8 + r3 = r3 | r4; // 9 + r1 = r1 | r2; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r6 = r6 | r2; // 12 + r2 = r2 * r5; // 13 + r1 = r1 ^ dataset[r2 & MASK]; // 14 + r7 = rotl_imm(r7, 1u); // 15 + { uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + r7 = r7 ^ dataset[r4 & MASK]; // 17 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 + r4 = r0 * r2 + r4; // 19 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20 + r5 = r5 ^ r7; // 21 + r2 = mulhi(r2, r5); // 22 + r3 = r3 ^ dataset[r7 & MASK]; // 23 + r7 = mulhi(r7, r3); // 24 + r5 = r5 | r4; // 25 + r4 = r5 * r2 + r4; // 26 + r5 = r5 * r1; // 27 + r6 = mulhi(r6, r7); // 28 + r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29 + r6 = rotr_var(r6, r7); // 30 + r3 = r3 ^ dataset[r1 & MASK]; // 31 + { uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 + r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 + { uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + r0 = r0 * r3; // 35 + r2 = r2 ^ r5; // 36 + r4 = r4 ^ dataset[r0 & MASK]; // 37 + r1 = r3 * r5 + r1; // 38 + r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39 + r2 = rotr_var(r2, r5); // 40 + r3 = r3 * r2; // 41 + r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42 + r3 = r3 ^ r4; // 43 + r3 = r3 ^ dataset[r5 & MASK]; // 44 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45 + r7 = r7 ^ r1; // 46 + r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47 + r7 = mulhi(r7, r5); // 48 + r0 = r0 ^ dataset[r2 & MASK]; // 49 + r2 = r2 - r6; // 50 + r7 = r7 - r5; // 51 + r2 = r2 ^ r3; // 52 + r7 = r7 - r0; // 53 + r3 = r5 * r0 + r3; // 54 + r7 = r7 ^ r5; // 55 + r2 = r2 ^ dataset[r7 & MASK]; // 56 + r5 = r5 - r6; // 57 + r1 = r1 ^ dataset[r3 & MASK]; // 58 + { uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 + r4 = r4 - r6; // 60 + r1 = r1 * r2; // 61 + r3 = r6 * r0 + r3; // 62 + r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} diff --git a/proto-cuda/packs-readwidth/scr4k128/program_bound.metal b/proto-cuda/packs-readwidth/scr4k128/program_bound.metal new file mode 100644 index 000000000..3257d287e --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k128/program_bound.metal @@ -0,0 +1,128 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW. +kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + constant uint* initw [[buffer(3)]], + device uint* scratch [[buffer(4)]], + constant uint& groups [[buffer(5)]], + constant uint& salt [[buffer(6)]], + uint tid [[thread_position_in_grid]], + uint nthreads [[threads_per_grid]]) { + uint lane = tid & 31u; + uint warp_ = tid >> 5; + uint nwarps_ = nthreads >> 5; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; } + { uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; } + { uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; } + { uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; } + { uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; } + { uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; } + { uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; } + { uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1 + r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2 + r4 = r0 * r6 + r4; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r4 = r4 ^ dataset[r1 & MASK]; // 5 + r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7 + r7 = r7 ^ r5; // 8 + r3 = r3 | r4; // 9 + r1 = r1 | r2; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r6 = r6 | r2; // 12 + r2 = r2 * r5; // 13 + r1 = r1 ^ dataset[r2 & MASK]; // 14 + r7 = rotl_imm(r7, 1u); // 15 + { uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + r7 = r7 ^ dataset[r4 & MASK]; // 17 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 + r4 = r0 * r2 + r4; // 19 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20 + r5 = r5 ^ r7; // 21 + r2 = mulhi(r2, r5); // 22 + r3 = r3 ^ dataset[r7 & MASK]; // 23 + r7 = mulhi(r7, r3); // 24 + r5 = r5 | r4; // 25 + r4 = r5 * r2 + r4; // 26 + r5 = r5 * r1; // 27 + r6 = mulhi(r6, r7); // 28 + r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29 + r6 = rotr_var(r6, r7); // 30 + r3 = r3 ^ dataset[r1 & MASK]; // 31 + { uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 + r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 + { uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + r0 = r0 * r3; // 35 + r2 = r2 ^ r5; // 36 + r4 = r4 ^ dataset[r0 & MASK]; // 37 + r1 = r3 * r5 + r1; // 38 + r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39 + r2 = rotr_var(r2, r5); // 40 + r3 = r3 * r2; // 41 + r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42 + r3 = r3 ^ r4; // 43 + r3 = r3 ^ dataset[r5 & MASK]; // 44 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45 + r7 = r7 ^ r1; // 46 + r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47 + r7 = mulhi(r7, r5); // 48 + r0 = r0 ^ dataset[r2 & MASK]; // 49 + r2 = r2 - r6; // 50 + r7 = r7 - r5; // 51 + r2 = r2 ^ r3; // 52 + r7 = r7 - r0; // 53 + r3 = r5 * r0 + r3; // 54 + r7 = r7 ^ r5; // 55 + r2 = r2 ^ dataset[r7 & MASK]; // 56 + r5 = r5 - r6; // 57 + r1 = r1 ^ dataset[r3 & MASK]; // 58 + { uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 + r4 = r4 - r6; // 60 + r1 = r1 * r2; // 61 + r3 = r6 * r0 + r3; // 62 + r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} diff --git a/proto-cuda/packs-readwidth/scr2/vectors.h b/proto-cuda/packs-readwidth/scr4k128/vectors.h similarity index 60% rename from proto-cuda/packs-readwidth/scr2/vectors.h rename to proto-cuda/packs-readwidth/scr4k128/vectors.h index 820c17e88..81688d9cc 100644 --- a/proto-cuda/packs-readwidth/scr2/vectors.h +++ b/proto-cuda/packs-readwidth/scr4k128/vectors.h @@ -11,22 +11,22 @@ static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { { // base nonce 0 - 0x2273e2732203e32aull, 0xa513d354bd107990ull, 0xe005b7515054c85full, 0x18a61b37b30cd1fbull, 0xa21d5b98e8d07e9cull, 0x5c24171a391d5ed0ull, 0x0f29e7583e1794b8ull, 0x7eca8a374d1a4f70ull, - 0x260ca011cd9ea10cull, 0xce1050798fce3d43ull, 0x9560d939dca19041ull, 0x8480a8440b80ecc3ull, 0xfaa99aac459b739eull, 0x7f083e72458e08abull, 0x78d876842f68672bull, 0x3b9bcf6275d3575cull, - 0x0256af61bdbf11b3ull, 0xefc6771cae646cbdull, 0xbc44f1c9f9f54d87ull, 0x6caedd783487eb7dull, 0x001b31fcb4fe0d4dull, 0x947a7ba1057e25b6ull, 0xb9e5a0204d68c22aull, 0x50489bed25d42661ull, - 0xb1019bff6d1057cdull, 0xd1442990562ce940ull, 0xcd986a47f98801dbull, 0x9c8796b6df23300full, 0xbd53ad05d2c877a9ull, 0xc95e863774a15b0aull, 0x132d8a91fb2fa67aull, 0x53dd38e8eadc8a24ull + 0xd48ade5043a1a440ull, 0x0236be04051c86abull, 0x4a368a9d6ce0d6eaull, 0x58518f225df7cd40ull, 0xa82cf7b3417df902ull, 0x4f2722d25e5aa401ull, 0xf94296c41a24bdb4ull, 0x4fca15537288398bull, + 0x9ba276d89178f795ull, 0x7205ccbf3772e4e5ull, 0x4149e0fbf1c35bb9ull, 0x91d484090f0b09e0ull, 0xb95010c64c2d27dbull, 0x2b9fc3c52f733771ull, 0x1f2f6dd44045e10cull, 0xb78170533d4b16a6ull, + 0x679056d6c824110full, 0x0f81ba5b1476c314ull, 0xeb3d83ce6ccd9cd2ull, 0x370575fe0b0e7119ull, 0x7b4ae5a0f119315full, 0x12ceac820c840cfcull, 0xd190a33169dd6d61ull, 0xf7f90f768fc7ec83ull, + 0x90f11a170fce1e73ull, 0xca168b60b8a41d61ull, 0xab9da4f155dc3d4cull, 0x884d93725ea8fc2full, 0x202bf861848ba637ull, 0x7d508d345e67589eull, 0x8dfcbd9365f96ddaull, 0x5e8315d79af30be3ull }, { // base nonce 4096 - 0x78c93312a03fb0eeull, 0x184fea638ec9b5fbull, 0x5687e8dcc4301dbfull, 0xed02c94f23681dfcull, 0x326d70162241ff6dull, 0x452017eb4ed2dfcfull, 0xc10b0e016f1e28c9ull, 0x691ce0cecf2a99baull, - 0x6c9506f34e0e63ceull, 0x447a98c2b7fdfa40ull, 0x07486b0e4055b2c9ull, 0x41781460bd47fd5cull, 0x01db316e35198291ull, 0xccd7e727f139a880ull, 0xdd7bd9efd16bf21cull, 0x8285d37966656366ull, - 0x383deade15fe0ecbull, 0x5fd64f5873c8e324ull, 0xad584cb6839c5e1dull, 0xbb842707fb5e9460ull, 0x4e8bc8f87978fcbdull, 0x18eb56f4a1fae881ull, 0x4c3b731a6b0c47a1ull, 0xda52cf9d69b252ebull, - 0xb5ff19b2b3eeb13eull, 0xe2595cea2afe42ddull, 0x3ff108424c9e6e38ull, 0x3a8a9e1995f359caull, 0x6a6b1da662cf2126ull, 0x54e684c127bb181full, 0x2018caa81f1a7d50ull, 0x33e94d2c92d9d148ull + 0x0925cd0a405af0f8ull, 0x9eeae6619738a6a0ull, 0x69d83343d36fa441ull, 0xef6e0dda67f22db7ull, 0xf5a5baddc99fc6e8ull, 0x048243c3a33d6313ull, 0xa2ed984433185d72ull, 0x5ce6444f5132231eull, + 0xf9c92e489ee1479bull, 0x13df97d418a1bb1dull, 0x54d8a4aa14eb5bf2ull, 0xc93ebe91c3aa1860ull, 0x12e1ca6f27af870bull, 0xa37cd8c938ec675bull, 0x0085b0d9144040d0ull, 0x389ce17c45d36ec5ull, + 0xc82d6694437e2f54ull, 0xf5a7b7357bc34eb6ull, 0x5e8e9c4cdaedb41dull, 0x18bff888b957603aull, 0x661b918790cf28f7ull, 0xf1411538bea6c80full, 0x0b3f28dd15dd2a0full, 0x076cd4e3230c6857ull, + 0x2cf5028d8fe8a18full, 0x270327a0333a8520ull, 0x53461f279f163486ull, 0x834a13f157378136ull, 0x5f8609aa7fbda1aeull, 0xd57be373dec55a73ull, 0xdf6b4c6134656905ull, 0x607a5b7e2a33765eull }, { // base nonce 1000000 - 0x04a41389bf3dfd3dull, 0xc509164def9207dfull, 0x4a8ffdbdf46e429dull, 0xff13bf0dc1b39aebull, 0xb852acc8e24133d7ull, 0x4bdd991ae56252acull, 0xa7739e74b3a054e9ull, 0xb4e36218d4b45fdcull, - 0x8bbd323155f5edc5ull, 0xb7b56a90659e7fd2ull, 0xdff7c495b7027480ull, 0xffa8adb5c0302b06ull, 0xe97d7967d89a5672ull, 0x0d0c2d4e6493926eull, 0xe9a5cda333cf2043ull, 0xdc95256d0986e5d8ull, - 0xd0dc211b811d6843ull, 0x68dfa3d0fb9a569bull, 0xa9e0028dfd9178c0ull, 0x4a36ca1fc40b20a9ull, 0xe7c765c5a735294bull, 0xf08954b015cb2628ull, 0xc69ee66ecf2740c5ull, 0xe3d01e899e46b089ull, - 0xc3558c74159c8603ull, 0x4c7aeb196bd01b04ull, 0x13c17119385f1910ull, 0xda7fca98e0989b8aull, 0x6f95baf340817945ull, 0x1af52756fd3afcabull, 0xb8eefc370bbe7e4bull, 0x94a55055b48bd4dbull + 0x16b8e21167f3437cull, 0xfd28ad2d0f75a03cull, 0xf8e70bdb604cfff7ull, 0xeba037043c5ece4bull, 0xa2cb7d31f4d25317ull, 0xc3e8b85a50bdab1dull, 0xd7bd65a4353eab2eull, 0x23540281fae8cce3ull, + 0x37f4deac1a67cc5full, 0xd482f81bec2535a5ull, 0xc18f3f46f812b870ull, 0x582514aab0cf566dull, 0xb3b1a7424a758ac6ull, 0x83bbf70ed4151fa4ull, 0x72e2fed205f44a00ull, 0x2f81d15c1a8e17feull, + 0xbe7875e7927ca851ull, 0x1a72a20d292cc17bull, 0x859dd2c75675a04bull, 0xe12711e4d81b1e04ull, 0xfafefd6afdda6b35ull, 0x30ebb4d12f5cf4e3ull, 0x6d4aae24eee724a2ull, 0x317d83e64bf5e9c6ull, + 0x2beba8ecb8b0b26eull, 0x5cf2eacb7a58bd99ull, 0x2d56441aac88a037ull, 0x01708cc58adfeb95ull, 0xb3bc095b95418a2full, 0xf804c273322638e1ull, 0x9d89d48818056f24ull, 0xe07cbfd9aa54ccf1ull } }; diff --git a/proto-cuda/packs-readwidth/scr4/vectors.json b/proto-cuda/packs-readwidth/scr4k128/vectors.json similarity index 65% rename from proto-cuda/packs-readwidth/scr4/vectors.json rename to proto-cuda/packs-readwidth/scr4k128/vectors.json index a0f951221..c394f4c58 100644 --- a/proto-cuda/packs-readwidth/scr4/vectors.json +++ b/proto-cuda/packs-readwidth/scr4k128/vectors.json @@ -8,22 +8,22 @@ "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", "warps": [ {"base_nonce": 0, "expected": [ - "0x66cffcc97c46e625", "0xbc9019f8df50fbfd", "0x65629c90dde6016e", "0xd647a41effa03d3b", "0x86da3b6bbd751b99", "0x6ccf4240a0fb2d19", "0xeb39a1e06f17378c", "0x2ea6b349b289fb10", - "0x6067211e6c220500", "0x6e6095dfedd1360f", "0xbd1190d8b50e1b48", "0x216dc72a0c08d5b5", "0x5be1f8c080836b0c", "0x2a32932a5953ed73", "0xcc2a3d68be83c802", "0xe46daca15338278f", - "0xb2de43b96761e459", "0x9004acd06588cbea", "0x6a9a3543cf93004f", "0xff956d859cb6e408", "0x4397ec6e3c5fb045", "0x521dea569cd481d5", "0x89832b34108759f0", "0xf66e393836ffe4ea", - "0xb4e39af6c40ea2f4", "0x3adc22085dd8d648", "0x27efe270958bbfbb", "0x6c80be0e8dca60d8", "0xa0afbc6a60260d59", "0x5d9a257fb9189537", "0xeb837aeef55dc3ed", "0xd381174dc14f8951" + "0xd48ade5043a1a440", "0x0236be04051c86ab", "0x4a368a9d6ce0d6ea", "0x58518f225df7cd40", "0xa82cf7b3417df902", "0x4f2722d25e5aa401", "0xf94296c41a24bdb4", "0x4fca15537288398b", + "0x9ba276d89178f795", "0x7205ccbf3772e4e5", "0x4149e0fbf1c35bb9", "0x91d484090f0b09e0", "0xb95010c64c2d27db", "0x2b9fc3c52f733771", "0x1f2f6dd44045e10c", "0xb78170533d4b16a6", + "0x679056d6c824110f", "0x0f81ba5b1476c314", "0xeb3d83ce6ccd9cd2", "0x370575fe0b0e7119", "0x7b4ae5a0f119315f", "0x12ceac820c840cfc", "0xd190a33169dd6d61", "0xf7f90f768fc7ec83", + "0x90f11a170fce1e73", "0xca168b60b8a41d61", "0xab9da4f155dc3d4c", "0x884d93725ea8fc2f", "0x202bf861848ba637", "0x7d508d345e67589e", "0x8dfcbd9365f96dda", "0x5e8315d79af30be3" ]}, {"base_nonce": 4096, "expected": [ - "0xfb1f61aaeaeaef94", "0x184c2160963d8b57", "0x42a053c628625778", "0xeaad0e41c4770812", "0x1d6d389ceb462ce1", "0x4a639827672bdbd4", "0x857e42aa5a42dd6f", "0xb4ef399e5339979c", - "0x497a29225b099233", "0x71d8b42862d81954", "0x0af995663313bf04", "0xf436fd126619d7a1", "0x199e4e3333cff269", "0x64077952f3775768", "0x51af1d126c5e8388", "0xffbaf44fe6b15cfd", - "0xfc8fed86ecae34a7", "0x4cb548616f7a7d6b", "0xc21d938c8b5bef35", "0x34789cbdd7088f71", "0xacb099a2c207d891", "0xfe1902d162374413", "0x26f7831c28f4020b", "0xdf5192952b4af6b0", - "0xecab61fe88dbaff4", "0x941c491f7fdb86e5", "0x2b1900c53f746e77", "0x8c40507b1caffeb2", "0x7532a1ec2b9169ef", "0x1cf399b0c8bfb520", "0xdf003d2bb8a2cc0c", "0x4da853307fc977a9" + "0x0925cd0a405af0f8", "0x9eeae6619738a6a0", "0x69d83343d36fa441", "0xef6e0dda67f22db7", "0xf5a5baddc99fc6e8", "0x048243c3a33d6313", "0xa2ed984433185d72", "0x5ce6444f5132231e", + "0xf9c92e489ee1479b", "0x13df97d418a1bb1d", "0x54d8a4aa14eb5bf2", "0xc93ebe91c3aa1860", "0x12e1ca6f27af870b", "0xa37cd8c938ec675b", "0x0085b0d9144040d0", "0x389ce17c45d36ec5", + "0xc82d6694437e2f54", "0xf5a7b7357bc34eb6", "0x5e8e9c4cdaedb41d", "0x18bff888b957603a", "0x661b918790cf28f7", "0xf1411538bea6c80f", "0x0b3f28dd15dd2a0f", "0x076cd4e3230c6857", + "0x2cf5028d8fe8a18f", "0x270327a0333a8520", "0x53461f279f163486", "0x834a13f157378136", "0x5f8609aa7fbda1ae", "0xd57be373dec55a73", "0xdf6b4c6134656905", "0x607a5b7e2a33765e" ]}, {"base_nonce": 1000000, "expected": [ - "0x3d094bd04694b96f", "0xaecbd76cecd1a20a", "0xbcb86febe56b17fe", "0x98082b557ba97517", "0xbb5f94108888564b", "0xea3284877a30fc87", "0xc608fa4d5a8bb2ad", "0x946c721c511e0729", - "0x46c6eba292083aed", "0x936cb97231eb6795", "0xb1413c434c712cbb", "0xedfd554d3948c1bd", "0xa8a20cbef2faccd5", "0x5fe39d756cadbcad", "0x208b2627380791fe", "0xf52f9374ce480218", - "0xb9db7cd8814eb29e", "0xf32ed2192b5a8719", "0x4f1b06a054940aef", "0x406df498e4365eb5", "0x1982075caad345ef", "0x590f725623dbbbbd", "0xa26d9192dedfefa5", "0x36219ec00da18980", - "0x5361d0dcb0f8b1a3", "0x35bdefa2fbb5ffc3", "0xba4c2a4e473a9c80", "0x107d9d3030f8b9d3", "0xa8bb094266d6b987", "0x86164fdfbb1426e8", "0xa6e8cb895021cbbd", "0xfe12809e9d99a243" + "0x16b8e21167f3437c", "0xfd28ad2d0f75a03c", "0xf8e70bdb604cfff7", "0xeba037043c5ece4b", "0xa2cb7d31f4d25317", "0xc3e8b85a50bdab1d", "0xd7bd65a4353eab2e", "0x23540281fae8cce3", + "0x37f4deac1a67cc5f", "0xd482f81bec2535a5", "0xc18f3f46f812b870", "0x582514aab0cf566d", "0xb3b1a7424a758ac6", "0x83bbf70ed4151fa4", "0x72e2fed205f44a00", "0x2f81d15c1a8e17fe", + "0xbe7875e7927ca851", "0x1a72a20d292cc17b", "0x859dd2c75675a04b", "0xe12711e4d81b1e04", "0xfafefd6afdda6b35", "0x30ebb4d12f5cf4e3", "0x6d4aae24eee724a2", "0x317d83e64bf5e9c6", + "0x2beba8ecb8b0b26e", "0x5cf2eacb7a58bd99", "0x2d56441aac88a037", "0x01708cc58adfeb95", "0xb3bc095b95418a2f", "0xf804c273322638e1", "0x9d89d48818056f24", "0xe07cbfd9aa54ccf1" ]} ], "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], diff --git a/proto-cuda/packs-readwidth/scr4k32/kernel.cl b/proto-cuda/packs-readwidth/scr4k32/kernel.cl new file mode 100644 index 000000000..00f785968 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k32/kernel.cl @@ -0,0 +1,291 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d))) +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) { + uint lane = (uint)get_global_id(0) & 31u; + uint warp_ = (uint)get_global_id(0) >> 5; + uint nwarps_ = (uint)get_global_size(0) >> 5; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + r7 = r7 ^ ds[r2 & mask]; // 4 load + r4 = r4 ^ ds[r1 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = mul_hi(r2, r5); // 22 mulhi + r3 = r3 ^ ds[r7 & mask]; // 23 load + r7 = mul_hi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = mul_hi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + r4 = r4 ^ ds[r0 & mask]; // 37 load + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + r3 = r3 ^ ds[r5 & mask]; // 44 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = mul_hi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif diff --git a/proto-cuda/packs-readwidth/scr4k32/kernel.cu b/proto-cuda/packs-readwidth/scr4k32/kernel.cu new file mode 100644 index 000000000..d5a4717ed --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k32/kernel.cu @@ -0,0 +1,177 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). +// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. +#include +#include +#include "program.h" +#include "memhard.h" + +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { + uint32_t x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item. +// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host. +__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + uint32_t t = blockIdx.x * blockDim.x + threadIdx.x; + if (t < nItems) { + uint32_t s[16]; + mh_item(cache, t, s); + uint32_t* d = ds + (size_t)t * 16u; + for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} + +// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every +// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a +// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, uint32_t* scratch, uint32_t groups, uint32_t salt) { + uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; + uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; + uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { + uint32_t gid = g_ * 32u + lane; + uint32_t gbase = baseNonce + g_ * 32u; + uint32_t tag = salt + g_; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + r7 = r7 ^ ds[r2 & mask]; // 4 load + r4 = r4 ^ ds[r1 & mask]; // 5 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = __umulhi(r2, r5); // 22 mulhi + r3 = r3 ^ ds[r7 & mask]; // 23 load + r7 = __umulhi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = __umulhi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + r4 = r4 ^ ds[r0 & mask]; // 37 load + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + r3 = r3 ^ ds[r5 & mask]; // 44 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = __umulhi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; + } +} + +// Host-side launch wrappers. Declared in program.h, called from host.cu. +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_cache_fill<<>>(cache, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + if (nItems == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nItems + block - 1u) / block; + igneum_build<<>>(ds, cache, nItems); + return cudaGetLastError(); +} + +// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it). +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) { + if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue; + igneum_hash<<>>(ds, out, baseNonce, mask, scratch, nonces / 32u, salt); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-readwidth/scr4k32/kernel_bound.cl b/proto-cuda/packs-readwidth/scr4k32/kernel_bound.cl new file mode 100644 index 000000000..46034e31b --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k32/kernel_bound.cl @@ -0,0 +1,393 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d))) +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) { + uint lane = (uint)get_global_id(0) & 31u; + uint warp_ = (uint)get_global_id(0) >> 5; + uint nwarps_ = (uint)get_global_size(0) >> 5; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + r7 = r7 ^ ds[r2 & mask]; // 4 load + r4 = r4 ^ ds[r1 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = mul_hi(r2, r5); // 22 mulhi + r3 = r3 ^ ds[r7 & mask]; // 23 load + r7 = mul_hi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = mul_hi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + r4 = r4 ^ ds[r0 & mask]; // 37 load + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + r3 = r3 ^ ds[r5 & mask]; // 44 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = mul_hi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif + +// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash. +IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global uint* scratch, uint groups, uint salt) { + uint lane = (uint)get_global_id(0) & 31u; + uint warp_ = (uint)get_global_id(0) >> 5; + uint nwarps_ = (uint)get_global_size(0) >> 5; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7]; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; } + { uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; } + { uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; } + { uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; } + { uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; } + { uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; } + { uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; } + { uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + r7 = r7 ^ ds[r2 & mask]; // 4 load + r4 = r4 ^ ds[r1 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = mul_hi(r2, r5); // 22 mulhi + r3 = r3 ^ ds[r7 & mask]; // 23 load + r7 = mul_hi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = mul_hi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + r4 = r4 ^ ds[r0 & mask]; // 37 load + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + r3 = r3 ^ ds[r5 & mask]; // 44 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = mul_hi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} diff --git a/proto-cuda/packs-readwidth/scr4k32/kernel_bound.cu b/proto-cuda/packs-readwidth/scr4k32/kernel_bound.cu new file mode 100644 index 000000000..dc5ec90df --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k32/kernel_bound.cu @@ -0,0 +1,136 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW. +// Host declarations (also in program_bound.h if present): +// struct IgneumInitWords { uint32_t w[8]; }; +// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, +// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps); +// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#include +#include +#include "program.h" + +struct IgneumInitWords { uint32_t w[8]; }; + +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } + +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, uint32_t* scratch, uint32_t groups, uint32_t salt) { + uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; + uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; + uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { + uint32_t gid = g_ * 32u + lane; + uint32_t gbase = baseNonce + g_ * 32u; + uint32_t tag = salt + g_; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; } + { uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; } + { uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; } + { uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; } + { uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; } + { uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; } + { uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; } + { uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; } + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + r7 = r7 ^ ds[r2 & mask]; // 4 load + r4 = r4 ^ ds[r1 & mask]; // 5 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = __umulhi(r2, r5); // 22 mulhi + r3 = r3 ^ ds[r7 & mask]; // 23 load + r7 = __umulhi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = __umulhi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + r4 = r4 ^ ds[r0 & mask]; // 37 load + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + r3 = r3 ^ ds[r5 & mask]; // 44 load + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = __umulhi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; + } +} + +// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it). +cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) { + if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue; + igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, scratch, nonces / 32u, salt); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-readwidth/scr4k32/memhard.h b/proto-cuda/packs-readwidth/scr4k32/memhard.h new file mode 100644 index 000000000..4803d8e40 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k32/memhard.h @@ -0,0 +1,108 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against. +// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference). +// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#if defined(__CUDACC__) +#define IGNEUM_HD __host__ __device__ __forceinline__ +#elif defined(_MSC_VER) && !defined(__cplusplus) +#define IGNEUM_HD static __inline +#else +#define IGNEUM_HD static inline +#endif +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) { + for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint32_t r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint32_t r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } diff --git a/proto-cuda/packs-readwidth/scr4k32/memhard.metal b/proto-cuda/packs-readwidth/scr4k32/memhard.metal new file mode 100644 index 000000000..241866369 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k32/memhard.metal @@ -0,0 +1,106 @@ +#include +using namespace metal; +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +inline void mh_chacha_block(const thread uint* x, thread uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +inline void mh_cache_segment(device uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +inline void mh_mixer(thread uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +inline void mh_item(device const uint* cache, uint t, thread uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// One thread per segment (2^16 threads). +kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + mh_cache_segment(cache, gid); +} +// One thread per 64-byte item (dataset words / 16 threads). +kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]], + uint gid [[thread_position_in_grid]]) { + uint s[16]; + mh_item(cache, gid, s); + device uint* d = dataset + gid * 16u; + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; +} diff --git a/proto-cuda/packs-readwidth/scr4/program.h b/proto-cuda/packs-readwidth/scr4k32/program.h similarity index 91% rename from proto-cuda/packs-readwidth/scr4/program.h rename to proto-cuda/packs-readwidth/scr4k32/program.h index f8f7a6823..869816f8c 100644 --- a/proto-cuda/packs-readwidth/scr4/program.h +++ b/proto-cuda/packs-readwidth/scr4k32/program.h @@ -15,7 +15,7 @@ #define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" #define IGNEUM_GENERATOR 2 #define IGNEUM_PROGRAM_ATTEMPT 0 -#define IGNEUM_PROGRAM_ID 0x2f098ee568f386f5ull +#define IGNEUM_PROGRAM_ID 0xe0c4a4d155ce1befull #define IGNEUM_DAY_STRING "2026-10-03" #define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" #define IGNEUM_DAY0 0x3067619fu @@ -30,20 +30,20 @@ #define IGNEUM_OP_MIX "load=12 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 scratch=4 shfl=4 rotr=2 rotl=1" // Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads // the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x. -#define IGNEUM_LOAD_CLASS "scr4" +#define IGNEUM_LOAD_CLASS "scr4k32" #define IGNEUM_LOAD_SLOTS 16 #define IGNEUM_LOAD_MIX { 100, 0, 0 } #define IGNEUM_LOAD_WIDTH_COUNTS { 12, 0, 0 } // loads of 4, 16, 64 bytes per program #define IGNEUM_BYTES_PER_HASH 384 #define IGNEUM_FOLD_ROT 11 #define IGNEUM_FOLD_MUL 0x9e3779b1u -// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch, +// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch, // groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit). #define IGNEUM_PERSISTENT_WARPS 1 #define IGNEUM_SCRATCH_OPS 4 // scratch read-modify-writes per program (32 per hash) -#define IGNEUM_SCRATCH_SLOTS 2048u -#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u -#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u +#define IGNEUM_SCRATCH_SLOTS 64u +#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u +#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u // 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) #define IGNEUM_DATASET_MODE 1 diff --git a/proto-cuda/packs-readwidth/scr4k32/program.json b/proto-cuda/packs-readwidth/scr4k32/program.json new file mode 100644 index 000000000..9bd6cf0f5 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k32/program.json @@ -0,0 +1,130 @@ +{ + "format": "igneum-program-pack-3", + "generator": 2, + "attempt": 0, + "program_id": "0xe0c4a4d155ce1bef", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-genesis", + "seed_bytes": "69676e65756d2d67656e65736973", + "seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 128, + "load_class": "scr4k32", + "load_slots": 16, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [12, 0, 0], + "bytes_per_hash": 384, + "scratch_ops_per_hash": 32, + "scratch_kib_per_warp": 32, + "scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)", + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "op_mix": {"load": 12, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "scratch": 4, "shfl": 4, "rotr": 2, "rotl": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "2026-10-03", + "day_bytes": "6461792f323032362d31302d3033", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0x3067619f", + "d1": "0x3c269176", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "instructions": [ + {"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1}, + {"i": 1, "op": "add", "dst": 1, "src": 7, "src2": 2, "imm": "0x42da7657", "imm2": "0xc3bd2355", "rot": 25, "bit": 4, "mask": 16, "width": 1}, + {"i": 2, "op": "add", "dst": 2, "src": 3, "src2": 0, "imm": "0x61f0b51c", "imm2": "0x2735a174", "rot": 4, "bit": 26, "mask": 2, "width": 1}, + {"i": 3, "op": "mad", "dst": 4, "src": 0, "src2": 6, "imm": "0x679648a8", "imm2": "0x3044ba32", "rot": 31, "bit": 31, "mask": 4, "width": 1}, + {"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1}, + {"i": 5, "op": "load", "dst": 4, "src": 1, "src2": 2, "imm": "0x987c017a", "imm2": "0xf4d60559", "rot": 2, "bit": 0, "mask": 4, "width": 1}, + {"i": 6, "op": "shfl", "dst": 6, "src": 3, "src2": 7, "imm": "0x6ea7b2df", "imm2": "0x9fce5071", "rot": 7, "bit": 15, "mask": 4, "width": 1}, + {"i": 7, "op": "shfl", "dst": 1, "src": 5, "src2": 1, "imm": "0x26a2ecde", "imm2": "0xfec6ad22", "rot": 15, "bit": 11, "mask": 8, "width": 1}, + {"i": 8, "op": "xor", "dst": 7, "src": 5, "src2": 2, "imm": "0xbe4b445c", "imm2": "0x17a5a9c7", "rot": 8, "bit": 8, "mask": 1, "width": 1}, + {"i": 9, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1}, + {"i": 10, "op": "or", "dst": 1, "src": 2, "src2": 3, "imm": "0x4e7dc10d", "imm2": "0x196d165c", "rot": 14, "bit": 27, "mask": 16, "width": 1}, + {"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 1, "imm": "0xc5c3b55d", "imm2": "0xec061424", "rot": 26, "bit": 27, "mask": 8, "width": 1}, + {"i": 12, "op": "or", "dst": 6, "src": 2, "src2": 3, "imm": "0x306542fe", "imm2": "0x1bb1b429", "rot": 31, "bit": 0, "mask": 2, "width": 1}, + {"i": 13, "op": "mul", "dst": 2, "src": 5, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1}, + {"i": 14, "op": "load", "dst": 1, "src": 2, "src2": 5, "imm": "0x028b4d37", "imm2": "0x7bbd78ea", "rot": 15, "bit": 2, "mask": 8, "width": 1}, + {"i": 15, "op": "rotl", "dst": 7, "src": 6, "src2": 6, "imm": "0x5c88a1a7", "imm2": "0x5c628769", "rot": 1, "bit": 3, "mask": 8, "width": 1}, + {"i": 16, "op": "scratch", "dst": 3, "src": 6, "src2": 7, "imm": "0xbac2ae81", "imm2": "0xcbbc7bdb", "rot": 18, "bit": 8, "mask": 8, "width": 1}, + {"i": 17, "op": "load", "dst": 7, "src": 4, "src2": 2, "imm": "0xe8ab93e9", "imm2": "0xa00de107", "rot": 2, "bit": 1, "mask": 16, "width": 1}, + {"i": 18, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1}, + {"i": 19, "op": "mad", "dst": 4, "src": 0, "src2": 2, "imm": "0x5fba7bc2", "imm2": "0xdf099cfb", "rot": 4, "bit": 15, "mask": 16, "width": 1}, + {"i": 20, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0x0a3056de", "imm2": "0x7f0c25c3", "rot": 27, "bit": 13, "mask": 8, "width": 1}, + {"i": 21, "op": "xor", "dst": 5, "src": 7, "src2": 4, "imm": "0xbd066e1d", "imm2": "0x6d3ddc5a", "rot": 2, "bit": 29, "mask": 1, "width": 1}, + {"i": 22, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1}, + {"i": 23, "op": "load", "dst": 3, "src": 7, "src2": 2, "imm": "0xc7fcfc8f", "imm2": "0x8528b94f", "rot": 17, "bit": 13, "mask": 4, "width": 1}, + {"i": 24, "op": "mulhi", "dst": 7, "src": 3, "src2": 5, "imm": "0xd91641e8", "imm2": "0xaf77faf2", "rot": 22, "bit": 21, "mask": 1, "width": 1}, + {"i": 25, "op": "or", "dst": 5, "src": 4, "src2": 0, "imm": "0x84c03868", "imm2": "0xf6c691b7", "rot": 29, "bit": 14, "mask": 8, "width": 1}, + {"i": 26, "op": "mad", "dst": 4, "src": 5, "src2": 2, "imm": "0x3bb2b6ba", "imm2": "0x49d95fd5", "rot": 1, "bit": 5, "mask": 8, "width": 1}, + {"i": 27, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1}, + {"i": 28, "op": "mulhi", "dst": 6, "src": 7, "src2": 6, "imm": "0xd69c4715", "imm2": "0xe0ebc4ce", "rot": 29, "bit": 2, "mask": 8, "width": 1}, + {"i": 29, "op": "add", "dst": 6, "src": 1, "src2": 2, "imm": "0x3b2d2124", "imm2": "0x187a9128", "rot": 1, "bit": 9, "mask": 16, "width": 1}, + {"i": 30, "op": "rotr", "dst": 6, "src": 7, "src2": 0, "imm": "0x5c64a589", "imm2": "0x61c9a38d", "rot": 17, "bit": 21, "mask": 16, "width": 1}, + {"i": 31, "op": "load", "dst": 3, "src": 1, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1}, + {"i": 32, "op": "scratch", "dst": 1, "src": 0, "src2": 7, "imm": "0xcc7972c4", "imm2": "0xad098d15", "rot": 30, "bit": 21, "mask": 8, "width": 1}, + {"i": 33, "op": "add", "dst": 0, "src": 4, "src2": 4, "imm": "0x2c35699f", "imm2": "0x351dde38", "rot": 21, "bit": 18, "mask": 4, "width": 1}, + {"i": 34, "op": "scratch", "dst": 0, "src": 2, "src2": 3, "imm": "0xfae8902b", "imm2": "0x5cd8306f", "rot": 5, "bit": 28, "mask": 16, "width": 1}, + {"i": 35, "op": "mul", "dst": 0, "src": 3, "src2": 1, "imm": "0x4fa3f3db", "imm2": "0xdbf37e75", "rot": 7, "bit": 18, "mask": 4, "width": 1}, + {"i": 36, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1}, + {"i": 37, "op": "load", "dst": 4, "src": 0, "src2": 0, "imm": "0x04cc1d55", "imm2": "0x35c52d04", "rot": 11, "bit": 14, "mask": 2, "width": 1}, + {"i": 38, "op": "mad", "dst": 1, "src": 3, "src2": 5, "imm": "0x3958f280", "imm2": "0x8713c7e1", "rot": 5, "bit": 23, "mask": 16, "width": 1}, + {"i": 39, "op": "add", "dst": 0, "src": 3, "src2": 3, "imm": "0xa907b90b", "imm2": "0x1b053acf", "rot": 30, "bit": 25, "mask": 16, "width": 1}, + {"i": 40, "op": "rotr", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1}, + {"i": 41, "op": "mul", "dst": 3, "src": 2, "src2": 4, "imm": "0x49087d74", "imm2": "0x6348b489", "rot": 17, "bit": 9, "mask": 16, "width": 1}, + {"i": 42, "op": "add", "dst": 1, "src": 5, "src2": 1, "imm": "0xa32e000c", "imm2": "0x6058c2e3", "rot": 25, "bit": 20, "mask": 8, "width": 1}, + {"i": 43, "op": "xor", "dst": 3, "src": 4, "src2": 2, "imm": "0x3dad0eb6", "imm2": "0xb97578cb", "rot": 3, "bit": 27, "mask": 1, "width": 1}, + {"i": 44, "op": "load", "dst": 3, "src": 5, "src2": 7, "imm": "0x374aec92", "imm2": "0x626f11df", "rot": 20, "bit": 18, "mask": 8, "width": 1}, + {"i": 45, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1}, + {"i": 46, "op": "xor", "dst": 7, "src": 1, "src2": 0, "imm": "0xef6ac348", "imm2": "0x963bb7e6", "rot": 26, "bit": 3, "mask": 8, "width": 1}, + {"i": 47, "op": "add", "dst": 0, "src": 3, "src2": 0, "imm": "0x838b5065", "imm2": "0x36360066", "rot": 3, "bit": 31, "mask": 4, "width": 1}, + {"i": 48, "op": "mulhi", "dst": 7, "src": 5, "src2": 0, "imm": "0x8458f7ac", "imm2": "0xc1c15026", "rot": 27, "bit": 15, "mask": 8, "width": 1}, + {"i": 49, "op": "load", "dst": 0, "src": 2, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1}, + {"i": 50, "op": "sub", "dst": 2, "src": 6, "src2": 0, "imm": "0x2baec8c9", "imm2": "0x4390f156", "rot": 3, "bit": 12, "mask": 8, "width": 1}, + {"i": 51, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x19234061", "imm2": "0xe84dfade", "rot": 4, "bit": 19, "mask": 1, "width": 1}, + {"i": 52, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0xdc2cd71e", "imm2": "0x1b5d334b", "rot": 9, "bit": 8, "mask": 8, "width": 1}, + {"i": 53, "op": "sub", "dst": 7, "src": 0, "src2": 4, "imm": "0x605c31ec", "imm2": "0x9923ff88", "rot": 28, "bit": 25, "mask": 4, "width": 1}, + {"i": 54, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1}, + {"i": 55, "op": "xor", "dst": 7, "src": 5, "src2": 5, "imm": "0xad7493e7", "imm2": "0x3e400372", "rot": 13, "bit": 8, "mask": 1, "width": 1}, + {"i": 56, "op": "load", "dst": 2, "src": 7, "src2": 1, "imm": "0x87e933c9", "imm2": "0x8c854c1b", "rot": 17, "bit": 3, "mask": 8, "width": 1}, + {"i": 57, "op": "sub", "dst": 5, "src": 6, "src2": 5, "imm": "0x11be3bc9", "imm2": "0xbbaa8e24", "rot": 6, "bit": 5, "mask": 16, "width": 1}, + {"i": 58, "op": "load", "dst": 1, "src": 3, "src2": 2, "imm": "0xa732351a", "imm2": "0xc01349cd", "rot": 14, "bit": 17, "mask": 16, "width": 1}, + {"i": 59, "op": "scratch", "dst": 1, "src": 4, "src2": 0, "imm": "0xb20547b2", "imm2": "0xc94655de", "rot": 27, "bit": 30, "mask": 1, "width": 1}, + {"i": 60, "op": "sub", "dst": 4, "src": 6, "src2": 7, "imm": "0x67cf904c", "imm2": "0x6873b216", "rot": 27, "bit": 7, "mask": 16, "width": 1}, + {"i": 61, "op": "mul", "dst": 1, "src": 2, "src2": 7, "imm": "0x93ab0bf4", "imm2": "0x96158375", "rot": 14, "bit": 0, "mask": 16, "width": 1}, + {"i": 62, "op": "mad", "dst": 3, "src": 6, "src2": 0, "imm": "0x41a443a3", "imm2": "0xe69d7919", "rot": 9, "bit": 0, "mask": 16, "width": 1}, + {"i": 63, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x2fe0e98b", "imm2": "0xc88e2942", "rot": 5, "bit": 16, "mask": 16, "width": 1} + ] +} diff --git a/proto-cuda/packs-readwidth/scr4k32/program.metal b/proto-cuda/packs-readwidth/scr4k32/program.metal new file mode 100644 index 000000000..ecfc2be8d --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k32/program.metal @@ -0,0 +1,126 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +kernel void igneum_hash(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + device uint* scratch [[buffer(3)]], + constant uint& groups [[buffer(4)]], + constant uint& salt [[buffer(5)]], + uint tid [[thread_position_in_grid]], + uint nthreads [[threads_per_grid]]) { + uint lane = tid & 31u; + uint warp_ = tid >> 5; + uint nwarps_ = nthreads >> 5; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; } + { uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; } + { uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; } + { uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; } + { uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; } + { uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; } + { uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; } + { uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1 + r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2 + r4 = r0 * r6 + r4; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r4 = r4 ^ dataset[r1 & MASK]; // 5 + r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7 + r7 = r7 ^ r5; // 8 + r3 = r3 | r4; // 9 + r1 = r1 | r2; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r6 = r6 | r2; // 12 + r2 = r2 * r5; // 13 + r1 = r1 ^ dataset[r2 & MASK]; // 14 + r7 = rotl_imm(r7, 1u); // 15 + { uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + r7 = r7 ^ dataset[r4 & MASK]; // 17 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 + r4 = r0 * r2 + r4; // 19 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20 + r5 = r5 ^ r7; // 21 + r2 = mulhi(r2, r5); // 22 + r3 = r3 ^ dataset[r7 & MASK]; // 23 + r7 = mulhi(r7, r3); // 24 + r5 = r5 | r4; // 25 + r4 = r5 * r2 + r4; // 26 + r5 = r5 * r1; // 27 + r6 = mulhi(r6, r7); // 28 + r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29 + r6 = rotr_var(r6, r7); // 30 + r3 = r3 ^ dataset[r1 & MASK]; // 31 + { uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 + r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 + { uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + r0 = r0 * r3; // 35 + r2 = r2 ^ r5; // 36 + r4 = r4 ^ dataset[r0 & MASK]; // 37 + r1 = r3 * r5 + r1; // 38 + r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39 + r2 = rotr_var(r2, r5); // 40 + r3 = r3 * r2; // 41 + r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42 + r3 = r3 ^ r4; // 43 + r3 = r3 ^ dataset[r5 & MASK]; // 44 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45 + r7 = r7 ^ r1; // 46 + r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47 + r7 = mulhi(r7, r5); // 48 + r0 = r0 ^ dataset[r2 & MASK]; // 49 + r2 = r2 - r6; // 50 + r7 = r7 - r5; // 51 + r2 = r2 ^ r3; // 52 + r7 = r7 - r0; // 53 + r3 = r5 * r0 + r3; // 54 + r7 = r7 ^ r5; // 55 + r2 = r2 ^ dataset[r7 & MASK]; // 56 + r5 = r5 - r6; // 57 + r1 = r1 ^ dataset[r3 & MASK]; // 58 + { uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 + r4 = r4 - r6; // 60 + r1 = r1 * r2; // 61 + r3 = r6 * r0 + r3; // 62 + r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} diff --git a/proto-cuda/packs-readwidth/scr4k32/program_bound.metal b/proto-cuda/packs-readwidth/scr4k32/program_bound.metal new file mode 100644 index 000000000..cf840249d --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k32/program_bound.metal @@ -0,0 +1,128 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW. +kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + constant uint* initw [[buffer(3)]], + device uint* scratch [[buffer(4)]], + constant uint& groups [[buffer(5)]], + constant uint& salt [[buffer(6)]], + uint tid [[thread_position_in_grid]], + uint nthreads [[threads_per_grid]]) { + uint lane = tid & 31u; + uint warp_ = tid >> 5; + uint nwarps_ = nthreads >> 5; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; } + { uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; } + { uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; } + { uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; } + { uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; } + { uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; } + { uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; } + { uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1 + r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2 + r4 = r0 * r6 + r4; // 3 + r7 = r7 ^ dataset[r2 & MASK]; // 4 + r4 = r4 ^ dataset[r1 & MASK]; // 5 + r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7 + r7 = r7 ^ r5; // 8 + r3 = r3 | r4; // 9 + r1 = r1 | r2; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r6 = r6 | r2; // 12 + r2 = r2 * r5; // 13 + r1 = r1 ^ dataset[r2 & MASK]; // 14 + r7 = rotl_imm(r7, 1u); // 15 + { uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + r7 = r7 ^ dataset[r4 & MASK]; // 17 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 + r4 = r0 * r2 + r4; // 19 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20 + r5 = r5 ^ r7; // 21 + r2 = mulhi(r2, r5); // 22 + r3 = r3 ^ dataset[r7 & MASK]; // 23 + r7 = mulhi(r7, r3); // 24 + r5 = r5 | r4; // 25 + r4 = r5 * r2 + r4; // 26 + r5 = r5 * r1; // 27 + r6 = mulhi(r6, r7); // 28 + r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29 + r6 = rotr_var(r6, r7); // 30 + r3 = r3 ^ dataset[r1 & MASK]; // 31 + { uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 + r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 + { uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + r0 = r0 * r3; // 35 + r2 = r2 ^ r5; // 36 + r4 = r4 ^ dataset[r0 & MASK]; // 37 + r1 = r3 * r5 + r1; // 38 + r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39 + r2 = rotr_var(r2, r5); // 40 + r3 = r3 * r2; // 41 + r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42 + r3 = r3 ^ r4; // 43 + r3 = r3 ^ dataset[r5 & MASK]; // 44 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45 + r7 = r7 ^ r1; // 46 + r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47 + r7 = mulhi(r7, r5); // 48 + r0 = r0 ^ dataset[r2 & MASK]; // 49 + r2 = r2 - r6; // 50 + r7 = r7 - r5; // 51 + r2 = r2 ^ r3; // 52 + r7 = r7 - r0; // 53 + r3 = r5 * r0 + r3; // 54 + r7 = r7 ^ r5; // 55 + r2 = r2 ^ dataset[r7 & MASK]; // 56 + r5 = r5 - r6; // 57 + r1 = r1 ^ dataset[r3 & MASK]; // 58 + { uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 + r4 = r4 - r6; // 60 + r1 = r1 * r2; // 61 + r3 = r6 * r0 + r3; // 62 + r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} diff --git a/proto-cuda/packs-readwidth/scr4k32/vectors.h b/proto-cuda/packs-readwidth/scr4k32/vectors.h new file mode 100644 index 000000000..e15e27475 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k32/vectors.h @@ -0,0 +1,57 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif + +#define IGNEUM_VEC_WARPS 3 +static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; +static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { + { // base nonce 0 + 0x62cab4be0ed880e1ull, 0x90842c849268cf52ull, 0x53d4d591f4a12749ull, 0x020429c3d1279eddull, 0x7086876f3a9183fbull, 0xc1f954e8065b6d02ull, 0x7550abd24b6ed9edull, 0x7dcc57f17b255fc9ull, + 0x01f16667b6ea326dull, 0x05907ad2b28423a5ull, 0x27e8aad8889ea702ull, 0xe93e61d2ff565fabull, 0x21aa77db95a8790full, 0xdc361b8011f7a395ull, 0xbabf590a4cf6dd58ull, 0x7a9b7eb46cf2b88dull, + 0xeb43f11d39490205ull, 0x3c2e3b2ed2b0c6fbull, 0x2bc2293577914bc8ull, 0x66cc059287de6cc0ull, 0x9a2d3a0f30169b20ull, 0xa5756e5027ec459full, 0x734a7fb8546a08b5ull, 0xbc59502ef67511d7ull, + 0x869d40616fa13209ull, 0x7e2713e23d7c3e06ull, 0x640f81856492a8d3ull, 0x1f7ce7b42962bc41ull, 0xc474fdd339993867ull, 0xb08d199c21d08397ull, 0x6b021b07d5dd9fdfull, 0x574adb548a4f3be8ull + }, + { // base nonce 4096 + 0x2956e7703c1553fcull, 0xe06e9c5dd64f0cffull, 0x41d967b788797c3eull, 0xc9beec7571ab7808ull, 0x0d0d99d51ac72942ull, 0xac60f79ae1bb46d6ull, 0xd9b9a509bc33d145ull, 0x512d444977d257f5ull, + 0x0e9759a10d773b28ull, 0xb270e841265b2b3dull, 0x90b97771870e53dcull, 0xf8b0bacb1ead0c1bull, 0x32165ca85108736cull, 0x5a907f0cb371d6d1ull, 0x89d4ccf9b6323847ull, 0x495e23339db371dbull, + 0xa50d559aa7911894ull, 0xbe561e2c2e64f0ffull, 0x862b141ae3b898eaull, 0x69b52c3068f0544aull, 0x2769f3b4051f9e80ull, 0xb679a28f140a5ccaull, 0x037d194732dce935ull, 0xec9f1e85406dbee9ull, + 0x65a0d3f10857795dull, 0x5da0b4908b5cda66ull, 0x1cbdf4f47dad39e4ull, 0x5472317d40d55545ull, 0x24ec2fb5eeff7691ull, 0x4c56104e2454b9beull, 0x8b896d9e85dbf491ull, 0xe80882d5975e09ecull + }, + { // base nonce 1000000 + 0xa417c0e0494f5f0dull, 0x44cfa8bbf55cb40bull, 0xee534b970664a105ull, 0x2814b1857db92d67ull, 0xd358f7e47b35ff5full, 0xa08faa47e58221c3ull, 0xbb76559b9a4a447bull, 0xd438a3fd1fa2976eull, + 0xa08d0e2c88abe900ull, 0x58b3c3ab097c416dull, 0x705de177cf28ccdcull, 0x263f35e27d8cf3aaull, 0xa2c304ccfeb9ae9bull, 0x470ba6ea4e8f661aull, 0xa59e5f33cd8613d9ull, 0xdb887848353dc91cull, + 0xd948d1c36b6a98e1ull, 0xd78806062c54882aull, 0x0744e029194938aaull, 0x42b613ec3d9074c6ull, 0x88c75044753d1496ull, 0x940239d09cbd80e1ull, 0x4534dd536f53adf5ull, 0x6f44b9bd4d6564deull, + 0xb6d8142c422857d5ull, 0x3b61f0c20b8fb04bull, 0x17eaf3a88b49c9fdull, 0xec536015d760eb1eull, 0xf37e9ad57045cc05ull, 0x588b808bc29ab6caull, 0x7cfa7ae3c6e483c2ull, 0x71e3cc45c07b530cull + } +}; + +// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455). +static const uint32_t IGNEUM_DS_HEAD[16] = { + 0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu, + 0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du +}; +static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u; +static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u; +// 64 sampled dataset words (index, value) computed on the Mac. +#define IGNEUM_DS_SAMPLES 64 +static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = { + 59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u +}; +static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = { + 0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u +}; +// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words. +static const uint32_t IGNEUM_CACHE_HEAD[16] = { + 0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u, + 0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u +}; +static const uint32_t IGNEUM_CACHE_LAST[16] = { + 0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du, + 0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu +}; +static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull; diff --git a/proto-cuda/packs-readwidth/scr4k32/vectors.json b/proto-cuda/packs-readwidth/scr4k32/vectors.json new file mode 100644 index 000000000..4b970a994 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr4k32/vectors.json @@ -0,0 +1,36 @@ +{ + "seed": "igneum-genesis", + "day": "2026-10-03", + "dataset_mode": "memory-hard", + "dataset_log2_words": 28, + "mask": "0x0fffffff", + "lanes": 32, + "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", + "warps": [ + {"base_nonce": 0, "expected": [ + "0x62cab4be0ed880e1", "0x90842c849268cf52", "0x53d4d591f4a12749", "0x020429c3d1279edd", "0x7086876f3a9183fb", "0xc1f954e8065b6d02", "0x7550abd24b6ed9ed", "0x7dcc57f17b255fc9", + "0x01f16667b6ea326d", "0x05907ad2b28423a5", "0x27e8aad8889ea702", "0xe93e61d2ff565fab", "0x21aa77db95a8790f", "0xdc361b8011f7a395", "0xbabf590a4cf6dd58", "0x7a9b7eb46cf2b88d", + "0xeb43f11d39490205", "0x3c2e3b2ed2b0c6fb", "0x2bc2293577914bc8", "0x66cc059287de6cc0", "0x9a2d3a0f30169b20", "0xa5756e5027ec459f", "0x734a7fb8546a08b5", "0xbc59502ef67511d7", + "0x869d40616fa13209", "0x7e2713e23d7c3e06", "0x640f81856492a8d3", "0x1f7ce7b42962bc41", "0xc474fdd339993867", "0xb08d199c21d08397", "0x6b021b07d5dd9fdf", "0x574adb548a4f3be8" + ]}, + {"base_nonce": 4096, "expected": [ + "0x2956e7703c1553fc", "0xe06e9c5dd64f0cff", "0x41d967b788797c3e", "0xc9beec7571ab7808", "0x0d0d99d51ac72942", "0xac60f79ae1bb46d6", "0xd9b9a509bc33d145", "0x512d444977d257f5", + "0x0e9759a10d773b28", "0xb270e841265b2b3d", "0x90b97771870e53dc", "0xf8b0bacb1ead0c1b", "0x32165ca85108736c", "0x5a907f0cb371d6d1", "0x89d4ccf9b6323847", "0x495e23339db371db", + "0xa50d559aa7911894", "0xbe561e2c2e64f0ff", "0x862b141ae3b898ea", "0x69b52c3068f0544a", "0x2769f3b4051f9e80", "0xb679a28f140a5cca", "0x037d194732dce935", "0xec9f1e85406dbee9", + "0x65a0d3f10857795d", "0x5da0b4908b5cda66", "0x1cbdf4f47dad39e4", "0x5472317d40d55545", "0x24ec2fb5eeff7691", "0x4c56104e2454b9be", "0x8b896d9e85dbf491", "0xe80882d5975e09ec" + ]}, + {"base_nonce": 1000000, "expected": [ + "0xa417c0e0494f5f0d", "0x44cfa8bbf55cb40b", "0xee534b970664a105", "0x2814b1857db92d67", "0xd358f7e47b35ff5f", "0xa08faa47e58221c3", "0xbb76559b9a4a447b", "0xd438a3fd1fa2976e", + "0xa08d0e2c88abe900", "0x58b3c3ab097c416d", "0x705de177cf28ccdc", "0x263f35e27d8cf3aa", "0xa2c304ccfeb9ae9b", "0x470ba6ea4e8f661a", "0xa59e5f33cd8613d9", "0xdb887848353dc91c", + "0xd948d1c36b6a98e1", "0xd78806062c54882a", "0x0744e029194938aa", "0x42b613ec3d9074c6", "0x88c75044753d1496", "0x940239d09cbd80e1", "0x4534dd536f53adf5", "0x6f44b9bd4d6564de", + "0xb6d8142c422857d5", "0x3b61f0c20b8fb04b", "0x17eaf3a88b49c9fd", "0xec536015d760eb1e", "0xf37e9ad57045cc05", "0x588b808bc29ab6ca", "0x7cfa7ae3c6e483c2", "0x71e3cc45c07b530c" + ]} + ], + "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], + "dataset_last_index": 268435455, + "dataset_last": "0xa33ada72", + "dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}], + "cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"], + "cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"], + "cache_fnv1a64": "0x48c4f5bf24166b2e" +} diff --git a/proto-cuda/packs-readwidth/scr8k128/kernel.cl b/proto-cuda/packs-readwidth/scr8k128/kernel.cl new file mode 100644 index 000000000..03840512b --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k128/kernel.cl @@ -0,0 +1,291 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d))) +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) { + uint lane = (uint)get_global_id(0) & 31u; + uint warp_ = (uint)get_global_id(0) >> 5; + uint nwarps_ = (uint)get_global_size(0) >> 5; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch + r4 = r4 ^ ds[r1 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = mul_hi(r2, r5); // 22 mulhi + { uint s_ = r7 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch + r7 = mul_hi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = mul_hi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + { uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + { uint s_ = r5 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = mul_hi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif diff --git a/proto-cuda/packs-readwidth/scr8k128/kernel.cu b/proto-cuda/packs-readwidth/scr8k128/kernel.cu new file mode 100644 index 000000000..9db090564 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k128/kernel.cu @@ -0,0 +1,177 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). +// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. +#include +#include +#include "program.h" +#include "memhard.h" + +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { + uint32_t x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item. +// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host. +__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + uint32_t t = blockIdx.x * blockDim.x + threadIdx.x; + if (t < nItems) { + uint32_t s[16]; + mh_item(cache, t, s); + uint32_t* d = ds + (size_t)t * 16u; + for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} + +// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every +// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a +// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, uint32_t* scratch, uint32_t groups, uint32_t salt) { + uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; + uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; + uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; + for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { + uint32_t gid = g_ * 32u + lane; + uint32_t gbase = baseNonce + g_ * 32u; + uint32_t tag = salt + g_; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + { uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch + r4 = r4 ^ ds[r1 & mask]; // 5 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = __umulhi(r2, r5); // 22 mulhi + { uint32_t s_ = r7 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch + r7 = __umulhi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = __umulhi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + { uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + { uint32_t s_ = r5 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = __umulhi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; + } +} + +// Host-side launch wrappers. Declared in program.h, called from host.cu. +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_cache_fill<<>>(cache, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + if (nItems == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nItems + block - 1u) / block; + igneum_build<<>>(ds, cache, nItems); + return cudaGetLastError(); +} + +// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it). +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) { + if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue; + igneum_hash<<>>(ds, out, baseNonce, mask, scratch, nonces / 32u, salt); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-readwidth/scr8k128/kernel_bound.cl b/proto-cuda/packs-readwidth/scr8k128/kernel_bound.cl new file mode 100644 index 000000000..9cb72a38f --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k128/kernel_bound.cl @@ -0,0 +1,393 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d))) +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) { + uint lane = (uint)get_global_id(0) & 31u; + uint warp_ = (uint)get_global_id(0) >> 5; + uint nwarps_ = (uint)get_global_size(0) >> 5; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch + r4 = r4 ^ ds[r1 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = mul_hi(r2, r5); // 22 mulhi + { uint s_ = r7 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch + r7 = mul_hi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = mul_hi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + { uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + { uint s_ = r5 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = mul_hi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif + +// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash. +IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global uint* scratch, uint groups, uint salt) { + uint lane = (uint)get_global_id(0) & 31u; + uint warp_ = (uint)get_global_id(0) >> 5; + uint nwarps_ = (uint)get_global_size(0) >> 5; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7]; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; } + { uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; } + { uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; } + { uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; } + { uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; } + { uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; } + { uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; } + { uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch + r4 = r4 ^ ds[r1 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = mul_hi(r2, r5); // 22 mulhi + { uint s_ = r7 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch + r7 = mul_hi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = mul_hi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + { uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + { uint s_ = r5 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = mul_hi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} diff --git a/proto-cuda/packs-readwidth/scr8k128/kernel_bound.cu b/proto-cuda/packs-readwidth/scr8k128/kernel_bound.cu new file mode 100644 index 000000000..4224a414c --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k128/kernel_bound.cu @@ -0,0 +1,136 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW. +// Host declarations (also in program_bound.h if present): +// struct IgneumInitWords { uint32_t w[8]; }; +// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, +// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps); +// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#include +#include +#include "program.h" + +struct IgneumInitWords { uint32_t w[8]; }; + +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } + +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, uint32_t* scratch, uint32_t groups, uint32_t salt) { + uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; + uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; + uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; + for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { + uint32_t gid = g_ * 32u + lane; + uint32_t gbase = baseNonce + g_ * 32u; + uint32_t tag = salt + g_; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; } + { uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; } + { uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; } + { uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; } + { uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; } + { uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; } + { uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; } + { uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; } + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + { uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch + r4 = r4 ^ ds[r1 & mask]; // 5 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = __umulhi(r2, r5); // 22 mulhi + { uint32_t s_ = r7 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch + r7 = __umulhi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = __umulhi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + { uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + { uint32_t s_ = r5 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = __umulhi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; + } +} + +// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it). +cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) { + if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue; + igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, scratch, nonces / 32u, salt); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-readwidth/scr8k128/memhard.h b/proto-cuda/packs-readwidth/scr8k128/memhard.h new file mode 100644 index 000000000..4803d8e40 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k128/memhard.h @@ -0,0 +1,108 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against. +// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference). +// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#if defined(__CUDACC__) +#define IGNEUM_HD __host__ __device__ __forceinline__ +#elif defined(_MSC_VER) && !defined(__cplusplus) +#define IGNEUM_HD static __inline +#else +#define IGNEUM_HD static inline +#endif +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) { + for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint32_t r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint32_t r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } diff --git a/proto-cuda/packs-readwidth/scr8k128/memhard.metal b/proto-cuda/packs-readwidth/scr8k128/memhard.metal new file mode 100644 index 000000000..241866369 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k128/memhard.metal @@ -0,0 +1,106 @@ +#include +using namespace metal; +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +inline void mh_chacha_block(const thread uint* x, thread uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +inline void mh_cache_segment(device uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +inline void mh_mixer(thread uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +inline void mh_item(device const uint* cache, uint t, thread uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// One thread per segment (2^16 threads). +kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + mh_cache_segment(cache, gid); +} +// One thread per 64-byte item (dataset words / 16 threads). +kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]], + uint gid [[thread_position_in_grid]]) { + uint s[16]; + mh_item(cache, gid, s); + device uint* d = dataset + gid * 16u; + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; +} diff --git a/proto-cuda/packs-readwidth/scr8k128/program.h b/proto-cuda/packs-readwidth/scr8k128/program.h new file mode 100644 index 000000000..2da41c27e --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k128/program.h @@ -0,0 +1,67 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Program metadata for host.cu plus the launch wrappers defined in kernel.cu. +// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#ifndef IGNEUM_NO_CUDA +#include +#endif + +#define IGNEUM_SEED_STRING "igneum-genesis" +#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" +#define IGNEUM_GENERATOR 2 +#define IGNEUM_PROGRAM_ATTEMPT 0 +#define IGNEUM_PROGRAM_ID 0xe0d1dcd155d90573ull +#define IGNEUM_DAY_STRING "2026-10-03" +#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" +#define IGNEUM_DAY0 0x3067619fu +#define IGNEUM_DAY1 0x3c269176u +#define IGNEUM_DATASET_LOG2 28 +#define IGNEUM_MASK 0x0fffffffu +#define IGNEUM_LANES 32 +#define IGNEUM_ITERATIONS 8 +#define IGNEUM_INSTR_COUNT 64 +#define IGNEUM_LOADS_PER_HASH 128 +#define IGNEUM_WIDE_LOADS_PER_HASH 0 +#define IGNEUM_OP_MIX "add=9 load=8 scratch=8 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 rotl=1" +// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads +// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x. +#define IGNEUM_LOAD_CLASS "scr8k128" +#define IGNEUM_LOAD_SLOTS 16 +#define IGNEUM_LOAD_MIX { 100, 0, 0 } +#define IGNEUM_LOAD_WIDTH_COUNTS { 8, 0, 0 } // loads of 4, 16, 64 bytes per program +#define IGNEUM_BYTES_PER_HASH 256 +#define IGNEUM_FOLD_ROT 11 +#define IGNEUM_FOLD_MUL 0x9e3779b1u +// Variant 5: persistent warps, a 128 KiB scratch per launched warp (the host launches N warps and passes scratch, +// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit). +#define IGNEUM_PERSISTENT_WARPS 1 +#define IGNEUM_SCRATCH_OPS 8 // scratch read-modify-writes per program (64 per hash) +#define IGNEUM_SCRATCH_SLOTS 256u +#define IGNEUM_SCRATCH_WORDS_PER_LANE 1024u +#define IGNEUM_SCRATCH_BYTES_PER_WARP 131072u +// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) +#define IGNEUM_DATASET_MODE 1 + +#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u } +#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u } +#define IGNEUM_CACHE_LOG2_WORDS 26 +#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6 +#define IGNEUM_CACHE_SEGMENTS 65536u +#define IGNEUM_ITEM_ROUNDS 8 +#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u } +#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u } +#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u } + +#ifndef IGNEUM_NO_CUDA +// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError(). +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments); +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems); +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt); +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#endif diff --git a/proto-cuda/packs-readwidth/scr8/program.json b/proto-cuda/packs-readwidth/scr8k128/program.json similarity index 98% rename from proto-cuda/packs-readwidth/scr8/program.json rename to proto-cuda/packs-readwidth/scr8k128/program.json index 25348b09f..8cbee7a2b 100644 --- a/proto-cuda/packs-readwidth/scr8/program.json +++ b/proto-cuda/packs-readwidth/scr8k128/program.json @@ -2,7 +2,7 @@ "format": "igneum-program-pack-3", "generator": 2, "attempt": 0, - "program_id": "0x2f0992e568f38dc1", + "program_id": "0xe0d1dcd155d90573", "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", "dataset_mode": "memory-hard", "seed": "igneum-genesis", @@ -15,13 +15,14 @@ "iterations": 8, "instruction_count": 64, "loads_per_hash": 128, - "load_class": "scr8", + "load_class": "scr8k128", "load_slots": 16, "load_mix_percent_4_16_64": [100, 0, 0], "load_width_counts_4_16_64": [8, 0, 0], "bytes_per_hash": 256, "scratch_ops_per_hash": 64, - "scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)", + "scratch_kib_per_warp": 128, + "scratch": "variant 5 (measurement only): persistent warps; a 128 KiB scratch per warp of 256 16-byte slots per lane (lane-major); slot = src & 0xff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)", "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", "op_mix": {"add": 9, "load": 8, "scratch": 8, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "rotl": 1}, "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", diff --git a/proto-cuda/packs-readwidth/scr8/program.metal b/proto-cuda/packs-readwidth/scr8k128/program.metal similarity index 56% rename from proto-cuda/packs-readwidth/scr8/program.metal rename to proto-cuda/packs-readwidth/scr8k128/program.metal index 73532c454..c30f95277 100644 --- a/proto-cuda/packs-readwidth/scr8/program.metal +++ b/proto-cuda/packs-readwidth/scr8k128/program.metal @@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) { return x; } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], uint lane = tid & 31u; uint warp_ = tid >> 5; uint nwarps_ = nthreads >> 5; - device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -58,7 +58,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1 r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2 r4 = r0 * r6 + r4; // 3 - { uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 + { uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 r4 = r4 ^ dataset[r1 & MASK]; // 5 r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6 r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7 @@ -70,14 +70,14 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], r2 = r2 * r5; // 13 r1 = r1 ^ dataset[r2 & MASK]; // 14 r7 = rotl_imm(r7, 1u); // 15 - { uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + { uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 r7 = r7 ^ dataset[r4 & MASK]; // 17 r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 r4 = r0 * r2 + r4; // 19 r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20 r5 = r5 ^ r7; // 21 r2 = mulhi(r2, r5); // 22 - { uint s_ = r7 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 + { uint s_ = r7 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 r7 = mulhi(r7, r3); // 24 r5 = r5 | r4; // 25 r4 = r5 * r2 + r4; // 26 @@ -86,19 +86,19 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29 r6 = rotr_var(r6, r7); // 30 r3 = r3 ^ dataset[r1 & MASK]; // 31 - { uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 + { uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 - { uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + { uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 r0 = r0 * r3; // 35 r2 = r2 ^ r5; // 36 - { uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 + { uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 r1 = r3 * r5 + r1; // 38 r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39 r2 = rotr_var(r2, r5); // 40 r3 = r3 * r2; // 41 r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42 r3 = r3 ^ r4; // 43 - { uint s_ = r5 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 + { uint s_ = r5 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45 r7 = r7 ^ r1; // 46 r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47 @@ -113,7 +113,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]], r2 = r2 ^ dataset[r7 & MASK]; // 56 r5 = r5 - r6; // 57 r1 = r1 ^ dataset[r3 & MASK]; // 58 - { uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 + { uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 r4 = r4 - r6; // 60 r1 = r1 * r2; // 61 r3 = r6 * r0 + r3; // 62 diff --git a/proto-cuda/packs-readwidth/scr8/program_bound.metal b/proto-cuda/packs-readwidth/scr8k128/program_bound.metal similarity index 57% rename from proto-cuda/packs-readwidth/scr8/program_bound.metal rename to proto-cuda/packs-readwidth/scr8k128/program_bound.metal index d260c34fe..665f2721e 100644 --- a/proto-cuda/packs-readwidth/scr8/program_bound.metal +++ b/proto-cuda/packs-readwidth/scr8k128/program_bound.metal @@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) { return x; } -// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of // 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not // this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } @@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], uint lane = tid & 31u; uint warp_ = tid >> 5; uint nwarps_ = nthreads >> 5; - device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u; for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { uint gid = g_ * 32u + lane; uint gbase = baseNonce + g_ * 32u; @@ -60,7 +60,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1 r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2 r4 = r0 * r6 + r4; // 3 - { uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 + { uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 r4 = r4 ^ dataset[r1 & MASK]; // 5 r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6 r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7 @@ -72,14 +72,14 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], r2 = r2 * r5; // 13 r1 = r1 ^ dataset[r2 & MASK]; // 14 r7 = rotl_imm(r7, 1u); // 15 - { uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + { uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 r7 = r7 ^ dataset[r4 & MASK]; // 17 r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 r4 = r0 * r2 + r4; // 19 r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20 r5 = r5 ^ r7; // 21 r2 = mulhi(r2, r5); // 22 - { uint s_ = r7 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 + { uint s_ = r7 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 r7 = mulhi(r7, r3); // 24 r5 = r5 | r4; // 25 r4 = r5 * r2 + r4; // 26 @@ -88,19 +88,19 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29 r6 = rotr_var(r6, r7); // 30 r3 = r3 ^ dataset[r1 & MASK]; // 31 - { uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 + { uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 - { uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + { uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 r0 = r0 * r3; // 35 r2 = r2 ^ r5; // 36 - { uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 + { uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 r1 = r3 * r5 + r1; // 38 r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39 r2 = rotr_var(r2, r5); // 40 r3 = r3 * r2; // 41 r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42 r3 = r3 ^ r4; // 43 - { uint s_ = r5 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 + { uint s_ = r5 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45 r7 = r7 ^ r1; // 46 r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47 @@ -115,7 +115,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], r2 = r2 ^ dataset[r7 & MASK]; // 56 r5 = r5 - r6; // 57 r1 = r1 ^ dataset[r3 & MASK]; // 58 - { uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 + { uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 r4 = r4 - r6; // 60 r1 = r1 * r2; // 61 r3 = r6 * r0 + r3; // 62 diff --git a/proto-cuda/packs-readwidth/scr8k128/vectors.h b/proto-cuda/packs-readwidth/scr8k128/vectors.h new file mode 100644 index 000000000..4555a7d44 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k128/vectors.h @@ -0,0 +1,57 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif + +#define IGNEUM_VEC_WARPS 3 +static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; +static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { + { // base nonce 0 + 0xe814771d17db365cull, 0xc4d4a4ae8b6033caull, 0xa4ea884e5e961196ull, 0x5a4f2e353f13502bull, 0xb5bfae78c5048e4cull, 0x75ed7ff2f8ccb9c1ull, 0xc9d37d26079b1916ull, 0xef3c29ccb9d46163ull, + 0xb4cae00d3e73ae8eull, 0x88bc44f26a90e913ull, 0xeb91c51e87da69f8ull, 0x2ee04eeac0ba97b2ull, 0x43e33706056bb735ull, 0x88acef8db41e6bbfull, 0x87295bf633750804ull, 0x4a8310fa3c5393f3ull, + 0x9c33366c7aa6a5a6ull, 0x79ba6d5674f78e5cull, 0x03168e07fb7ae416ull, 0xdd45a1b5f54270aeull, 0x0d0084aa74f95b51ull, 0xa04060d3f711930bull, 0xa080e7988516297full, 0xcefa3e2e8de2806full, + 0x95dc7a55e10c4010ull, 0x62e809b8cef37be6ull, 0x0366273048e795cfull, 0xc2b04c7deacb1dffull, 0xe83446f7db3a4686ull, 0xe6c7e6575c34651full, 0xee15067b419606d6ull, 0xfe67f3ae51faade1ull + }, + { // base nonce 4096 + 0xdde6034f4b5824c9ull, 0x07b807430ab9effbull, 0xa581d141cb4bc74aull, 0x0a1b4129b3690618ull, 0xb4d10af1daaec58bull, 0xcf0b63a9aa6b8a96ull, 0x07c20bd30e3eb88cull, 0x32ebffcaafa5df9eull, + 0xe1b528deb263ddc3ull, 0xaa6d2e1e7f45c995ull, 0x6017aaa938e837cfull, 0x23445a9b8c8e5addull, 0x024ebd232a344f41ull, 0x67aebe3e79435f84ull, 0xa7d0b7522e88814aull, 0x1d4d9633b57ac637ull, + 0x77f500325912fcbdull, 0x9f4bc5d04fbb13b9ull, 0xd7e081a23934d582ull, 0x10992aa1c8a93afeull, 0x1596ba0b47520be7ull, 0x344ed3c63b5a75bcull, 0xdb75c50a7a39c7beull, 0x0e1ccc942ec1fad7ull, + 0x81c9d97c4d9605c6ull, 0x1e60918f98df7dc9ull, 0x3e60e5b90d94fe34ull, 0xec7163b65fc01cfdull, 0xb0786922940f66e3ull, 0xb1d049c3e24a38e2ull, 0x5a6a7ac9a0c8ed50ull, 0x7bc50eab43e83a01ull + }, + { // base nonce 1000000 + 0x1ebe406e6227f5e9ull, 0xc9d07c89dd188990ull, 0x3346fbacbe00f719ull, 0x437f4d678259e06dull, 0xd664758bc7508b7cull, 0xa3428dfd2b480593ull, 0xdfbb1aca3c18aeb6ull, 0xe362f4d90b64ab9full, + 0xfaeac4630c51e291ull, 0x2b81c4de8eaa689aull, 0x661b54d4d8763782ull, 0x48839cd831ea402bull, 0xa98dbad17e5f3a49ull, 0x35161bdc5dc87db7ull, 0xcc503293dddde770ull, 0x8960f9fbf15a0e47ull, + 0x3d391f32613d80d5ull, 0xd3fb60a843c31905ull, 0x7f70cbe3a4be1f1eull, 0xb67d653a57d143c7ull, 0x08a9210687821d3dull, 0x54adfc465596fd4cull, 0x12cf1cdd36c931e4ull, 0x62fd59b6a002a406ull, + 0xfef506618454af46ull, 0x8ab9f4c86cfefbe0ull, 0xb54d7b600ee5cbdbull, 0x9400685a172c31ccull, 0x0cdc0ca4f87996dcull, 0x050be1f1631ab65full, 0x79884be8f2e9a1f2ull, 0x824d03dfe16bcb91ull + } +}; + +// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455). +static const uint32_t IGNEUM_DS_HEAD[16] = { + 0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu, + 0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du +}; +static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u; +static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u; +// 64 sampled dataset words (index, value) computed on the Mac. +#define IGNEUM_DS_SAMPLES 64 +static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = { + 59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u +}; +static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = { + 0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u +}; +// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words. +static const uint32_t IGNEUM_CACHE_HEAD[16] = { + 0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u, + 0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u +}; +static const uint32_t IGNEUM_CACHE_LAST[16] = { + 0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du, + 0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu +}; +static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull; diff --git a/proto-cuda/packs-readwidth/scr8k128/vectors.json b/proto-cuda/packs-readwidth/scr8k128/vectors.json new file mode 100644 index 000000000..6f5c09848 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k128/vectors.json @@ -0,0 +1,36 @@ +{ + "seed": "igneum-genesis", + "day": "2026-10-03", + "dataset_mode": "memory-hard", + "dataset_log2_words": 28, + "mask": "0x0fffffff", + "lanes": 32, + "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", + "warps": [ + {"base_nonce": 0, "expected": [ + "0xe814771d17db365c", "0xc4d4a4ae8b6033ca", "0xa4ea884e5e961196", "0x5a4f2e353f13502b", "0xb5bfae78c5048e4c", "0x75ed7ff2f8ccb9c1", "0xc9d37d26079b1916", "0xef3c29ccb9d46163", + "0xb4cae00d3e73ae8e", "0x88bc44f26a90e913", "0xeb91c51e87da69f8", "0x2ee04eeac0ba97b2", "0x43e33706056bb735", "0x88acef8db41e6bbf", "0x87295bf633750804", "0x4a8310fa3c5393f3", + "0x9c33366c7aa6a5a6", "0x79ba6d5674f78e5c", "0x03168e07fb7ae416", "0xdd45a1b5f54270ae", "0x0d0084aa74f95b51", "0xa04060d3f711930b", "0xa080e7988516297f", "0xcefa3e2e8de2806f", + "0x95dc7a55e10c4010", "0x62e809b8cef37be6", "0x0366273048e795cf", "0xc2b04c7deacb1dff", "0xe83446f7db3a4686", "0xe6c7e6575c34651f", "0xee15067b419606d6", "0xfe67f3ae51faade1" + ]}, + {"base_nonce": 4096, "expected": [ + "0xdde6034f4b5824c9", "0x07b807430ab9effb", "0xa581d141cb4bc74a", "0x0a1b4129b3690618", "0xb4d10af1daaec58b", "0xcf0b63a9aa6b8a96", "0x07c20bd30e3eb88c", "0x32ebffcaafa5df9e", + "0xe1b528deb263ddc3", "0xaa6d2e1e7f45c995", "0x6017aaa938e837cf", "0x23445a9b8c8e5add", "0x024ebd232a344f41", "0x67aebe3e79435f84", "0xa7d0b7522e88814a", "0x1d4d9633b57ac637", + "0x77f500325912fcbd", "0x9f4bc5d04fbb13b9", "0xd7e081a23934d582", "0x10992aa1c8a93afe", "0x1596ba0b47520be7", "0x344ed3c63b5a75bc", "0xdb75c50a7a39c7be", "0x0e1ccc942ec1fad7", + "0x81c9d97c4d9605c6", "0x1e60918f98df7dc9", "0x3e60e5b90d94fe34", "0xec7163b65fc01cfd", "0xb0786922940f66e3", "0xb1d049c3e24a38e2", "0x5a6a7ac9a0c8ed50", "0x7bc50eab43e83a01" + ]}, + {"base_nonce": 1000000, "expected": [ + "0x1ebe406e6227f5e9", "0xc9d07c89dd188990", "0x3346fbacbe00f719", "0x437f4d678259e06d", "0xd664758bc7508b7c", "0xa3428dfd2b480593", "0xdfbb1aca3c18aeb6", "0xe362f4d90b64ab9f", + "0xfaeac4630c51e291", "0x2b81c4de8eaa689a", "0x661b54d4d8763782", "0x48839cd831ea402b", "0xa98dbad17e5f3a49", "0x35161bdc5dc87db7", "0xcc503293dddde770", "0x8960f9fbf15a0e47", + "0x3d391f32613d80d5", "0xd3fb60a843c31905", "0x7f70cbe3a4be1f1e", "0xb67d653a57d143c7", "0x08a9210687821d3d", "0x54adfc465596fd4c", "0x12cf1cdd36c931e4", "0x62fd59b6a002a406", + "0xfef506618454af46", "0x8ab9f4c86cfefbe0", "0xb54d7b600ee5cbdb", "0x9400685a172c31cc", "0x0cdc0ca4f87996dc", "0x050be1f1631ab65f", "0x79884be8f2e9a1f2", "0x824d03dfe16bcb91" + ]} + ], + "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], + "dataset_last_index": 268435455, + "dataset_last": "0xa33ada72", + "dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}], + "cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"], + "cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"], + "cache_fnv1a64": "0x48c4f5bf24166b2e" +} diff --git a/proto-cuda/packs-readwidth/scr8k32/kernel.cl b/proto-cuda/packs-readwidth/scr8k32/kernel.cl new file mode 100644 index 000000000..54168b39f --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k32/kernel.cl @@ -0,0 +1,291 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d))) +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) { + uint lane = (uint)get_global_id(0) & 31u; + uint warp_ = (uint)get_global_id(0) >> 5; + uint nwarps_ = (uint)get_global_size(0) >> 5; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch + r4 = r4 ^ ds[r1 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = mul_hi(r2, r5); // 22 mulhi + { uint s_ = r7 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch + r7 = mul_hi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = mul_hi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + { uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + { uint s_ = r5 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = mul_hi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif diff --git a/proto-cuda/packs-readwidth/scr8k32/kernel.cu b/proto-cuda/packs-readwidth/scr8k32/kernel.cu new file mode 100644 index 000000000..b43200fdd --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k32/kernel.cu @@ -0,0 +1,177 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal). +// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC. +#include +#include +#include "program.h" +#include "memhard.h" + +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31. +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x. +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) { + uint32_t x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item. +// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host. +__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) { + uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x; + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + uint32_t t = blockIdx.x * blockDim.x + threadIdx.x; + if (t < nItems) { + uint32_t s[16]; + mh_item(cache, t, s); + uint32_t* d = ds + (size_t)t * 16u; + for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} + +// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every +// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a +// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid. +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, uint32_t* scratch, uint32_t groups, uint32_t salt) { + uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; + uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; + uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { + uint32_t gid = g_ * 32u + lane; + uint32_t gbase = baseNonce + g_ * 32u; + uint32_t tag = salt + g_; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + { uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch + r4 = r4 ^ ds[r1 & mask]; // 5 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = __umulhi(r2, r5); // 22 mulhi + { uint32_t s_ = r7 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch + r7 = __umulhi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = __umulhi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + { uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + { uint32_t s_ = r5 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = __umulhi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; + } +} + +// Host-side launch wrappers. Declared in program.h, called from host.cu. +cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) { + if (nSegments == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nSegments + block - 1u) / block; + igneum_cache_fill<<>>(cache, nSegments); + return cudaGetLastError(); +} + +cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) { + if (nItems == 0u) return cudaErrorInvalidValue; + uint32_t block = 256u; + uint32_t grid = (nItems + block - 1u) / block; + igneum_build<<>>(ds, cache, nItems); + return cudaGetLastError(); +} + +// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it). +cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) { + if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue; + igneum_hash<<>>(ds, out, baseNonce, mask, scratch, nonces / 32u, salt); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-readwidth/scr8k32/kernel_bound.cl b/proto-cuda/packs-readwidth/scr8k32/kernel_bound.cl new file mode 100644 index 000000000..922739e4a --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k32/kernel_bound.cl @@ -0,0 +1,393 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal). +// Built from source at runtime by proto-opencl/host.c, which passes these defines: +// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit) +// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default) +// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32 +// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition +// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units; +// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32. +#ifndef IGNEUM_GROUP +#define IGNEUM_GROUP 32 +#endif +#ifndef IGNEUM_EXCHANGE +#define IGNEUM_EXCHANGE 0 +#endif +#ifdef __OPENCL_VERSION__ +#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1))) +#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n] +#if IGNEUM_EXCHANGE == 1 +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif +#ifdef cl_khr_subgroup_shuffle +#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable +#endif +#elif IGNEUM_EXCHANGE == 2 +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#endif +#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d))) +#else +// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros. +#include "emu_opencl.h" +#endif + +#if IGNEUM_EXCHANGE == 1 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#elif IGNEUM_EXCHANGE == 2 +#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m)) +#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u) +#else +// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per +// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane +// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's +// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier. +#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; } +#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; } +#endif + +static inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32. +static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); } +// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x. +static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); } +static inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +static inline void mh_chacha_block(const uint* x, uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +static inline void mh_cache_segment(__global uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + __global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +static inline void mh_mixer(uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +static inline void mh_item(__global const uint* cache, uint t, uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + __global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item. +// The same constants as memhard.h in this pack (one emitter, three dialects). +__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) { + uint seg = (uint)get_global_id(0); + if (seg < nSegments) mh_cache_segment(cache, seg); +} +__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) { + uint t = (uint)get_global_id(0); + if (t < nItems) { + uint s[16]; + mh_item(cache, t, s); + __global uint* d = ds + ((ulong)t * 16u); + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; + } +} + +// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the +// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and +// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all). +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) { + uint lane = (uint)get_global_id(0) & 31u; + uint warp_ = (uint)get_global_id(0) >> 5; + uint nwarps_ = (uint)get_global_size(0) >> 5; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1] + { uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2] + { uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3] + { uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4] + { uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5] + { uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6] + { uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7] + { uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0] + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch + r4 = r4 ^ ds[r1 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = mul_hi(r2, r5); // 22 mulhi + { uint s_ = r7 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch + r7 = mul_hi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = mul_hi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + { uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + { uint s_ = r5 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = mul_hi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} + +#if IGNEUM_EXCHANGE != 0 +// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the +// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a +// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md. +IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) { + if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); } +} +#endif + +// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash. +IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global uint* scratch, uint groups, uint salt) { + uint lane = (uint)get_global_id(0) & 31u; + uint warp_ = (uint)get_global_id(0) >> 5; + uint nwarps_ = (uint)get_global_size(0) >> 5; + __global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint lid = (uint)get_local_id(0); + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7]; +#if IGNEUM_EXCHANGE == 0 + IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP); + uint xk = 0u; +#else + (void)lid; +#endif + { uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; } + { uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; } + { uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; } + { uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; } + { uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; } + { uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; } + { uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; } + { uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch + r4 = r4 ^ ds[r1 & mask]; // 5 load + { uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl + { uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + { uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + { uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = mul_hi(r2, r5); // 22 mulhi + { uint s_ = r7 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch + r7 = mul_hi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = mul_hi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + { uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + { uint s_ = r5 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = mul_hi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} diff --git a/proto-cuda/packs-readwidth/scr8k32/kernel_bound.cu b/proto-cuda/packs-readwidth/scr8k32/kernel_bound.cu new file mode 100644 index 000000000..ae117b313 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k32/kernel_bound.cu @@ -0,0 +1,136 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW. +// Host declarations (also in program_bound.h if present): +// struct IgneumInitWords { uint32_t w[8]; }; +// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, +// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps); +// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps); +#include +#include +#include "program.h" + +struct IgneumInitWords { uint32_t w[8]; }; + +__device__ __forceinline__ uint32_t splitmix32(uint32_t x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } +__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } + +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, uint32_t* scratch, uint32_t groups, uint32_t salt) { + uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u; + uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5; + uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5; + uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) { + uint32_t gid = g_ * 32u + lane; + uint32_t gbase = baseNonce + g_ * 32u; + uint32_t tag = salt + g_; + uint32_t nonce = baseNonce + gid; + uint32_t r0, r1, r2, r3, r4, r5, r6, r7; + { uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; } + { uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; } + { uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; } + { uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; } + { uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; } + { uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; } + { uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; } + { uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; } + + for (uint32_t it = 0u; it < 8u; ++it) { + uint32_t sel = r0; + r2 = r3 * r4 + r2; // 0 mad + r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add + r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add + r4 = r0 * r6 + r4; // 3 mad + { uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch + r4 = r4 ^ ds[r1 & mask]; // 5 load + r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl + r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl + r7 = r7 ^ r5; // 8 xor + r3 = r3 | r4; // 9 or + r1 = r1 | r2; // 10 or + r4 = r4 ^ ds[r3 & mask]; // 11 load + r6 = r6 | r2; // 12 or + r2 = r2 * r5; // 13 mul + r1 = r1 ^ ds[r2 & mask]; // 14 load + r7 = rotl_imm(r7, 1u); // 15 rotl + { uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch + r7 = r7 ^ ds[r4 & mask]; // 17 load + r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl + r4 = r0 * r2 + r4; // 19 mad + r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl + r5 = r5 ^ r7; // 21 xor + r2 = __umulhi(r2, r5); // 22 mulhi + { uint32_t s_ = r7 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch + r7 = __umulhi(r7, r3); // 24 mulhi + r5 = r5 | r4; // 25 or + r4 = r5 * r2 + r4; // 26 mad + r5 = r5 * r1; // 27 mul + r6 = __umulhi(r6, r7); // 28 mulhi + r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add + r6 = rotr_var(r6, r7); // 30 rotr + r3 = r3 ^ ds[r1 & mask]; // 31 load + { uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch + r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add + { uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch + r0 = r0 * r3; // 35 mul + r2 = r2 ^ r5; // 36 xor + { uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch + r1 = r3 * r5 + r1; // 38 mad + r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add + r2 = rotr_var(r2, r5); // 40 rotr + r3 = r3 * r2; // 41 mul + r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add + r3 = r3 ^ r4; // 43 xor + { uint32_t s_ = r5 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch + r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add + r7 = r7 ^ r1; // 46 xor + r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add + r7 = __umulhi(r7, r5); // 48 mulhi + r0 = r0 ^ ds[r2 & mask]; // 49 load + r2 = r2 - r6; // 50 sub + r7 = r7 - r5; // 51 sub + r2 = r2 ^ r3; // 52 xor + r7 = r7 - r0; // 53 sub + r3 = r5 * r0 + r3; // 54 mad + r7 = r7 ^ r5; // 55 xor + r2 = r2 ^ ds[r7 & mask]; // 56 load + r5 = r5 - r6; // 57 sub + r1 = r1 ^ ds[r3 & mask]; // 58 load + { uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch + r4 = r4 - r6; // 60 sub + r1 = r1 * r2; // 61 mul + r3 = r6 * r0 + r3; // 62 mad + r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add + } + uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo; + } +} + +// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it). +cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, + IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) { + if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue; + uint32_t block = 32u * blockWarps; + if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue; + igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw, scratch, nonces / 32u, salt); + return cudaGetLastError(); +} + +cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) { + cudaFuncAttributes attr; + cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound); + if (e != cudaSuccess) return e; + *numRegs = attr.numRegs; + return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0); +} diff --git a/proto-cuda/packs-readwidth/scr8k32/memhard.h b/proto-cuda/packs-readwidth/scr8k32/memhard.h new file mode 100644 index 000000000..4803d8e40 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k32/memhard.h @@ -0,0 +1,108 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against. +// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference). +// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C. +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif +#if defined(__CUDACC__) +#define IGNEUM_HD __host__ __device__ __forceinline__ +#elif defined(_MSC_VER) && !defined(__cplusplus) +#define IGNEUM_HD static __inline +#else +#define IGNEUM_HD static inline +#endif +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) { + for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint32_t r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) { + uint32_t prev[16]; uint32_t x[16]; uint32_t y[16]; + for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint32_t r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } diff --git a/proto-cuda/packs-readwidth/scr8k32/memhard.metal b/proto-cuda/packs-readwidth/scr8k32/memhard.metal new file mode 100644 index 000000000..241866369 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k32/memhard.metal @@ -0,0 +1,106 @@ +#include +using namespace metal; +// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines. +// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals. +#define MH_CACHE_LINE_MASK 0x003fffffu +#define MH_SEGMENT_LINES 64u +#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); } +inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site + +// y = ChaCha12 core(x) + x +inline void mh_chacha_block(const thread uint* x, thread uint* y) { + for (uint i = 0u; i < 16u; ++i) y[i] = x[i]; + for (uint r = 0u; r < 6u; ++r) { + MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u) + MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u) + MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u) + } + for (uint i = 0u; i < 16u; ++i) y[i] += x[i]; +} + +// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0. +inline void mh_cache_segment(device uint* cache, uint seg) { + uint prev[16]; uint x[16]; uint y[16]; + for (uint i = 0u; i < 16u; ++i) prev[i] = 0u; + for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) { + x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3]; + x[4] = 0x3067619fu ^ prev[4]; + x[5] = 0x3c269176u ^ prev[5]; + x[6] = 0x84a03b03u ^ prev[6]; + x[7] = 0xf8c63294u ^ prev[7]; + x[8] = 0xff977c5bu ^ prev[8]; + x[9] = 0xe60def3eu ^ prev[9]; + x[10] = 0x63630141u ^ prev[10]; + x[11] = 0xb8fbcb58u ^ prev[11]; + x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15]; + mh_chacha_block(x, y); + device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u); + for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; } + } +} + +// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations. +inline void mh_mixer(thread uint* s, uint rk) { + s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u; + s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu; + s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u; + s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu; + s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u; + s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u; + s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u; + s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u; + s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du; + s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u; + s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du; + s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu; + s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du; + s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu; + s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u; + s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u; + MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u) + MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u) + MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u) + MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u) +} + +// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer. +inline void mh_item(device const uint* cache, uint t, thread uint* s) { + s[0] = 0x3067619fu; + s[1] = 0x3c269176u; + s[2] = 0x84a03b03u; + s[3] = 0xf8c63294u; + s[4] = 0xff977c5bu; + s[5] = 0xe60def3eu; + s[6] = 0x63630141u; + s[7] = 0xb8fbcb58u; + s[8] = t * 0x42146205u + 0xbab68293u; + s[9] = t * 0x52cbe0fbu + 0xcc162340u; + s[10] = t * 0x7ecf4a03u + 0x6ce151ccu; + s[11] = t * 0x6728907fu + 0xe62b8997u; + s[12] = t * 0xd81d9751u + 0xc9c80297u; + s[13] = t * 0x132952c3u + 0xf74a1654u; + s[14] = t * 0xf60de277u + 0x3d704af5u; + s[15] = t * 0x05358035u + 0x3cf522b7u; + for (uint r = 0u; r < 8u; ++r) { + mh_mixer(s, 0x9E3779B9u * (r + 1u)); + device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); + for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i]; + } + mh_mixer(s, 0x9E3779B9u * 9u); +} +// dataset[w] without the dataset: derive item w >> 4 and take word w & 15. +inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; } + +// One thread per segment (2^16 threads). +kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) { + mh_cache_segment(cache, gid); +} +// One thread per 64-byte item (dataset words / 16 threads). +kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]], + uint gid [[thread_position_in_grid]]) { + uint s[16]; + mh_item(cache, gid, s); + device uint* d = dataset + gid * 16u; + for (uint i = 0u; i < 16u; ++i) d[i] = s[i]; +} diff --git a/proto-cuda/packs-readwidth/scr8/program.h b/proto-cuda/packs-readwidth/scr8k32/program.h similarity index 91% rename from proto-cuda/packs-readwidth/scr8/program.h rename to proto-cuda/packs-readwidth/scr8k32/program.h index 769c814b6..d201dd8b0 100644 --- a/proto-cuda/packs-readwidth/scr8/program.h +++ b/proto-cuda/packs-readwidth/scr8k32/program.h @@ -15,7 +15,7 @@ #define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973" #define IGNEUM_GENERATOR 2 #define IGNEUM_PROGRAM_ATTEMPT 0 -#define IGNEUM_PROGRAM_ID 0x2f0992e568f38dc1ull +#define IGNEUM_PROGRAM_ID 0xe0d27cd155da1553ull #define IGNEUM_DAY_STRING "2026-10-03" #define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033" #define IGNEUM_DAY0 0x3067619fu @@ -30,20 +30,20 @@ #define IGNEUM_OP_MIX "add=9 load=8 scratch=8 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 rotl=1" // Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads // the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x. -#define IGNEUM_LOAD_CLASS "scr8" +#define IGNEUM_LOAD_CLASS "scr8k32" #define IGNEUM_LOAD_SLOTS 16 #define IGNEUM_LOAD_MIX { 100, 0, 0 } #define IGNEUM_LOAD_WIDTH_COUNTS { 8, 0, 0 } // loads of 4, 16, 64 bytes per program #define IGNEUM_BYTES_PER_HASH 256 #define IGNEUM_FOLD_ROT 11 #define IGNEUM_FOLD_MUL 0x9e3779b1u -// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch, +// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch, // groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit). #define IGNEUM_PERSISTENT_WARPS 1 #define IGNEUM_SCRATCH_OPS 8 // scratch read-modify-writes per program (64 per hash) -#define IGNEUM_SCRATCH_SLOTS 2048u -#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u -#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u +#define IGNEUM_SCRATCH_SLOTS 64u +#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u +#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u // 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h) #define IGNEUM_DATASET_MODE 1 diff --git a/proto-cuda/packs-readwidth/scr8k32/program.json b/proto-cuda/packs-readwidth/scr8k32/program.json new file mode 100644 index 000000000..3b0163f57 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k32/program.json @@ -0,0 +1,130 @@ +{ + "format": "igneum-program-pack-3", + "generator": 2, + "attempt": 0, + "program_id": "0xe0d27cd155da1553", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-genesis", + "seed_bytes": "69676e65756d2d67656e65736973", + "seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 128, + "load_class": "scr8k32", + "load_slots": 16, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [8, 0, 0], + "bytes_per_hash": 256, + "scratch_ops_per_hash": 64, + "scratch_kib_per_warp": 32, + "scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)", + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "op_mix": {"add": 9, "load": 8, "scratch": 8, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "rotl": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "2026-10-03", + "day_bytes": "6461792f323032362d31302d3033", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0x3067619f", + "d1": "0x3c269176", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "instructions": [ + {"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1}, + {"i": 1, "op": "add", "dst": 1, "src": 7, "src2": 2, "imm": "0x42da7657", "imm2": "0xc3bd2355", "rot": 25, "bit": 4, "mask": 16, "width": 1}, + {"i": 2, "op": "add", "dst": 2, "src": 3, "src2": 0, "imm": "0x61f0b51c", "imm2": "0x2735a174", "rot": 4, "bit": 26, "mask": 2, "width": 1}, + {"i": 3, "op": "mad", "dst": 4, "src": 0, "src2": 6, "imm": "0x679648a8", "imm2": "0x3044ba32", "rot": 31, "bit": 31, "mask": 4, "width": 1}, + {"i": 4, "op": "scratch", "dst": 7, "src": 2, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1}, + {"i": 5, "op": "load", "dst": 4, "src": 1, "src2": 2, "imm": "0x987c017a", "imm2": "0xf4d60559", "rot": 2, "bit": 0, "mask": 4, "width": 1}, + {"i": 6, "op": "shfl", "dst": 6, "src": 3, "src2": 7, "imm": "0x6ea7b2df", "imm2": "0x9fce5071", "rot": 7, "bit": 15, "mask": 4, "width": 1}, + {"i": 7, "op": "shfl", "dst": 1, "src": 5, "src2": 1, "imm": "0x26a2ecde", "imm2": "0xfec6ad22", "rot": 15, "bit": 11, "mask": 8, "width": 1}, + {"i": 8, "op": "xor", "dst": 7, "src": 5, "src2": 2, "imm": "0xbe4b445c", "imm2": "0x17a5a9c7", "rot": 8, "bit": 8, "mask": 1, "width": 1}, + {"i": 9, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1}, + {"i": 10, "op": "or", "dst": 1, "src": 2, "src2": 3, "imm": "0x4e7dc10d", "imm2": "0x196d165c", "rot": 14, "bit": 27, "mask": 16, "width": 1}, + {"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 1, "imm": "0xc5c3b55d", "imm2": "0xec061424", "rot": 26, "bit": 27, "mask": 8, "width": 1}, + {"i": 12, "op": "or", "dst": 6, "src": 2, "src2": 3, "imm": "0x306542fe", "imm2": "0x1bb1b429", "rot": 31, "bit": 0, "mask": 2, "width": 1}, + {"i": 13, "op": "mul", "dst": 2, "src": 5, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1}, + {"i": 14, "op": "load", "dst": 1, "src": 2, "src2": 5, "imm": "0x028b4d37", "imm2": "0x7bbd78ea", "rot": 15, "bit": 2, "mask": 8, "width": 1}, + {"i": 15, "op": "rotl", "dst": 7, "src": 6, "src2": 6, "imm": "0x5c88a1a7", "imm2": "0x5c628769", "rot": 1, "bit": 3, "mask": 8, "width": 1}, + {"i": 16, "op": "scratch", "dst": 3, "src": 6, "src2": 7, "imm": "0xbac2ae81", "imm2": "0xcbbc7bdb", "rot": 18, "bit": 8, "mask": 8, "width": 1}, + {"i": 17, "op": "load", "dst": 7, "src": 4, "src2": 2, "imm": "0xe8ab93e9", "imm2": "0xa00de107", "rot": 2, "bit": 1, "mask": 16, "width": 1}, + {"i": 18, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1}, + {"i": 19, "op": "mad", "dst": 4, "src": 0, "src2": 2, "imm": "0x5fba7bc2", "imm2": "0xdf099cfb", "rot": 4, "bit": 15, "mask": 16, "width": 1}, + {"i": 20, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0x0a3056de", "imm2": "0x7f0c25c3", "rot": 27, "bit": 13, "mask": 8, "width": 1}, + {"i": 21, "op": "xor", "dst": 5, "src": 7, "src2": 4, "imm": "0xbd066e1d", "imm2": "0x6d3ddc5a", "rot": 2, "bit": 29, "mask": 1, "width": 1}, + {"i": 22, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1}, + {"i": 23, "op": "scratch", "dst": 3, "src": 7, "src2": 2, "imm": "0xc7fcfc8f", "imm2": "0x8528b94f", "rot": 17, "bit": 13, "mask": 4, "width": 1}, + {"i": 24, "op": "mulhi", "dst": 7, "src": 3, "src2": 5, "imm": "0xd91641e8", "imm2": "0xaf77faf2", "rot": 22, "bit": 21, "mask": 1, "width": 1}, + {"i": 25, "op": "or", "dst": 5, "src": 4, "src2": 0, "imm": "0x84c03868", "imm2": "0xf6c691b7", "rot": 29, "bit": 14, "mask": 8, "width": 1}, + {"i": 26, "op": "mad", "dst": 4, "src": 5, "src2": 2, "imm": "0x3bb2b6ba", "imm2": "0x49d95fd5", "rot": 1, "bit": 5, "mask": 8, "width": 1}, + {"i": 27, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1}, + {"i": 28, "op": "mulhi", "dst": 6, "src": 7, "src2": 6, "imm": "0xd69c4715", "imm2": "0xe0ebc4ce", "rot": 29, "bit": 2, "mask": 8, "width": 1}, + {"i": 29, "op": "add", "dst": 6, "src": 1, "src2": 2, "imm": "0x3b2d2124", "imm2": "0x187a9128", "rot": 1, "bit": 9, "mask": 16, "width": 1}, + {"i": 30, "op": "rotr", "dst": 6, "src": 7, "src2": 0, "imm": "0x5c64a589", "imm2": "0x61c9a38d", "rot": 17, "bit": 21, "mask": 16, "width": 1}, + {"i": 31, "op": "load", "dst": 3, "src": 1, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1}, + {"i": 32, "op": "scratch", "dst": 1, "src": 0, "src2": 7, "imm": "0xcc7972c4", "imm2": "0xad098d15", "rot": 30, "bit": 21, "mask": 8, "width": 1}, + {"i": 33, "op": "add", "dst": 0, "src": 4, "src2": 4, "imm": "0x2c35699f", "imm2": "0x351dde38", "rot": 21, "bit": 18, "mask": 4, "width": 1}, + {"i": 34, "op": "scratch", "dst": 0, "src": 2, "src2": 3, "imm": "0xfae8902b", "imm2": "0x5cd8306f", "rot": 5, "bit": 28, "mask": 16, "width": 1}, + {"i": 35, "op": "mul", "dst": 0, "src": 3, "src2": 1, "imm": "0x4fa3f3db", "imm2": "0xdbf37e75", "rot": 7, "bit": 18, "mask": 4, "width": 1}, + {"i": 36, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1}, + {"i": 37, "op": "scratch", "dst": 4, "src": 0, "src2": 0, "imm": "0x04cc1d55", "imm2": "0x35c52d04", "rot": 11, "bit": 14, "mask": 2, "width": 1}, + {"i": 38, "op": "mad", "dst": 1, "src": 3, "src2": 5, "imm": "0x3958f280", "imm2": "0x8713c7e1", "rot": 5, "bit": 23, "mask": 16, "width": 1}, + {"i": 39, "op": "add", "dst": 0, "src": 3, "src2": 3, "imm": "0xa907b90b", "imm2": "0x1b053acf", "rot": 30, "bit": 25, "mask": 16, "width": 1}, + {"i": 40, "op": "rotr", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1}, + {"i": 41, "op": "mul", "dst": 3, "src": 2, "src2": 4, "imm": "0x49087d74", "imm2": "0x6348b489", "rot": 17, "bit": 9, "mask": 16, "width": 1}, + {"i": 42, "op": "add", "dst": 1, "src": 5, "src2": 1, "imm": "0xa32e000c", "imm2": "0x6058c2e3", "rot": 25, "bit": 20, "mask": 8, "width": 1}, + {"i": 43, "op": "xor", "dst": 3, "src": 4, "src2": 2, "imm": "0x3dad0eb6", "imm2": "0xb97578cb", "rot": 3, "bit": 27, "mask": 1, "width": 1}, + {"i": 44, "op": "scratch", "dst": 3, "src": 5, "src2": 7, "imm": "0x374aec92", "imm2": "0x626f11df", "rot": 20, "bit": 18, "mask": 8, "width": 1}, + {"i": 45, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1}, + {"i": 46, "op": "xor", "dst": 7, "src": 1, "src2": 0, "imm": "0xef6ac348", "imm2": "0x963bb7e6", "rot": 26, "bit": 3, "mask": 8, "width": 1}, + {"i": 47, "op": "add", "dst": 0, "src": 3, "src2": 0, "imm": "0x838b5065", "imm2": "0x36360066", "rot": 3, "bit": 31, "mask": 4, "width": 1}, + {"i": 48, "op": "mulhi", "dst": 7, "src": 5, "src2": 0, "imm": "0x8458f7ac", "imm2": "0xc1c15026", "rot": 27, "bit": 15, "mask": 8, "width": 1}, + {"i": 49, "op": "load", "dst": 0, "src": 2, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1}, + {"i": 50, "op": "sub", "dst": 2, "src": 6, "src2": 0, "imm": "0x2baec8c9", "imm2": "0x4390f156", "rot": 3, "bit": 12, "mask": 8, "width": 1}, + {"i": 51, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x19234061", "imm2": "0xe84dfade", "rot": 4, "bit": 19, "mask": 1, "width": 1}, + {"i": 52, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0xdc2cd71e", "imm2": "0x1b5d334b", "rot": 9, "bit": 8, "mask": 8, "width": 1}, + {"i": 53, "op": "sub", "dst": 7, "src": 0, "src2": 4, "imm": "0x605c31ec", "imm2": "0x9923ff88", "rot": 28, "bit": 25, "mask": 4, "width": 1}, + {"i": 54, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1}, + {"i": 55, "op": "xor", "dst": 7, "src": 5, "src2": 5, "imm": "0xad7493e7", "imm2": "0x3e400372", "rot": 13, "bit": 8, "mask": 1, "width": 1}, + {"i": 56, "op": "load", "dst": 2, "src": 7, "src2": 1, "imm": "0x87e933c9", "imm2": "0x8c854c1b", "rot": 17, "bit": 3, "mask": 8, "width": 1}, + {"i": 57, "op": "sub", "dst": 5, "src": 6, "src2": 5, "imm": "0x11be3bc9", "imm2": "0xbbaa8e24", "rot": 6, "bit": 5, "mask": 16, "width": 1}, + {"i": 58, "op": "load", "dst": 1, "src": 3, "src2": 2, "imm": "0xa732351a", "imm2": "0xc01349cd", "rot": 14, "bit": 17, "mask": 16, "width": 1}, + {"i": 59, "op": "scratch", "dst": 1, "src": 4, "src2": 0, "imm": "0xb20547b2", "imm2": "0xc94655de", "rot": 27, "bit": 30, "mask": 1, "width": 1}, + {"i": 60, "op": "sub", "dst": 4, "src": 6, "src2": 7, "imm": "0x67cf904c", "imm2": "0x6873b216", "rot": 27, "bit": 7, "mask": 16, "width": 1}, + {"i": 61, "op": "mul", "dst": 1, "src": 2, "src2": 7, "imm": "0x93ab0bf4", "imm2": "0x96158375", "rot": 14, "bit": 0, "mask": 16, "width": 1}, + {"i": 62, "op": "mad", "dst": 3, "src": 6, "src2": 0, "imm": "0x41a443a3", "imm2": "0xe69d7919", "rot": 9, "bit": 0, "mask": 16, "width": 1}, + {"i": 63, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x2fe0e98b", "imm2": "0xc88e2942", "rot": 5, "bit": 16, "mask": 16, "width": 1} + ] +} diff --git a/proto-cuda/packs-readwidth/scr8k32/program.metal b/proto-cuda/packs-readwidth/scr8k32/program.metal new file mode 100644 index 000000000..573e63513 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k32/program.metal @@ -0,0 +1,126 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +kernel void igneum_hash(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + device uint* scratch [[buffer(3)]], + constant uint& groups [[buffer(4)]], + constant uint& salt [[buffer(5)]], + uint tid [[thread_position_in_grid]], + uint nthreads [[threads_per_grid]]) { + uint lane = tid & 31u; + uint warp_ = tid >> 5; + uint nwarps_ = nthreads >> 5; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; } + { uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; } + { uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; } + { uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; } + { uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; } + { uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; } + { uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; } + { uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1 + r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2 + r4 = r0 * r6 + r4; // 3 + { uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 + r4 = r4 ^ dataset[r1 & MASK]; // 5 + r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7 + r7 = r7 ^ r5; // 8 + r3 = r3 | r4; // 9 + r1 = r1 | r2; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r6 = r6 | r2; // 12 + r2 = r2 * r5; // 13 + r1 = r1 ^ dataset[r2 & MASK]; // 14 + r7 = rotl_imm(r7, 1u); // 15 + { uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + r7 = r7 ^ dataset[r4 & MASK]; // 17 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 + r4 = r0 * r2 + r4; // 19 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20 + r5 = r5 ^ r7; // 21 + r2 = mulhi(r2, r5); // 22 + { uint s_ = r7 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 + r7 = mulhi(r7, r3); // 24 + r5 = r5 | r4; // 25 + r4 = r5 * r2 + r4; // 26 + r5 = r5 * r1; // 27 + r6 = mulhi(r6, r7); // 28 + r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29 + r6 = rotr_var(r6, r7); // 30 + r3 = r3 ^ dataset[r1 & MASK]; // 31 + { uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 + r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 + { uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + r0 = r0 * r3; // 35 + r2 = r2 ^ r5; // 36 + { uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 + r1 = r3 * r5 + r1; // 38 + r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39 + r2 = rotr_var(r2, r5); // 40 + r3 = r3 * r2; // 41 + r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42 + r3 = r3 ^ r4; // 43 + { uint s_ = r5 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45 + r7 = r7 ^ r1; // 46 + r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47 + r7 = mulhi(r7, r5); // 48 + r0 = r0 ^ dataset[r2 & MASK]; // 49 + r2 = r2 - r6; // 50 + r7 = r7 - r5; // 51 + r2 = r2 ^ r3; // 52 + r7 = r7 - r0; // 53 + r3 = r5 * r0 + r3; // 54 + r7 = r7 ^ r5; // 55 + r2 = r2 ^ dataset[r7 & MASK]; // 56 + r5 = r5 - r6; // 57 + r1 = r1 ^ dataset[r3 & MASK]; // 58 + { uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 + r4 = r4 - r6; // 60 + r1 = r1 * r2; // 61 + r3 = r6 * r0 + r3; // 62 + r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} diff --git a/proto-cuda/packs-readwidth/scr8k32/program_bound.metal b/proto-cuda/packs-readwidth/scr8k32/program_bound.metal new file mode 100644 index 000000000..e3ec2c039 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k32/program_bound.metal @@ -0,0 +1,128 @@ +#include +using namespace metal; + +#define MASK 0x0fffffffu +constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }; + +inline uint splitmix32(uint x) { + x ^= x >> 16; x *= 0x7feb352du; + x ^= x >> 15; x *= 0x846ca68bu; + x ^= x >> 16; + return x; +} +inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 +inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); } +inline uint ds_elem(uint i, uint d0, uint d1) { + uint x = i ^ d0; + x *= 0x9E3779B1u; x ^= x >> 15; + x += d1; + x *= 0x85EBCA77u; x ^= x >> 13; + x *= 0xC2B2AE3Du; x ^= x >> 16; + return x; +} + +// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of +// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not +// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill. +inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); } +// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW. +kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]], + device ulong* out [[buffer(1)]], + constant uint& baseNonce [[buffer(2)]], + constant uint* initw [[buffer(3)]], + device uint* scratch [[buffer(4)]], + constant uint& groups [[buffer(5)]], + constant uint& salt [[buffer(6)]], + uint tid [[thread_position_in_grid]], + uint nthreads [[threads_per_grid]]) { + uint lane = tid & 31u; + uint warp_ = tid >> 5; + uint nwarps_ = nthreads >> 5; + device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u; + for (uint g_ = warp_; g_ < groups; g_ += nwarps_) { + uint gid = g_ * 32u + lane; + uint gbase = baseNonce + g_ * 32u; + uint tag = salt + g_; + uint nonce = baseNonce + gid; + uint r0, r1, r2, r3, r4, r5, r6, r7; + { uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; } + { uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; } + { uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; } + { uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; } + { uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; } + { uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; } + { uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; } + { uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; } + + for (uint it = 0u; it < 8u; ++it) { + uint sel = r0; + r2 = r3 * r4 + r2; // 0 + r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1 + r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2 + r4 = r0 * r6 + r4; // 3 + { uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 + r4 = r4 ^ dataset[r1 & MASK]; // 5 + r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6 + r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7 + r7 = r7 ^ r5; // 8 + r3 = r3 | r4; // 9 + r1 = r1 | r2; // 10 + r4 = r4 ^ dataset[r3 & MASK]; // 11 + r6 = r6 | r2; // 12 + r2 = r2 * r5; // 13 + r1 = r1 ^ dataset[r2 & MASK]; // 14 + r7 = rotl_imm(r7, 1u); // 15 + { uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 + r7 = r7 ^ dataset[r4 & MASK]; // 17 + r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18 + r4 = r0 * r2 + r4; // 19 + r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20 + r5 = r5 ^ r7; // 21 + r2 = mulhi(r2, r5); // 22 + { uint s_ = r7 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 + r7 = mulhi(r7, r3); // 24 + r5 = r5 | r4; // 25 + r4 = r5 * r2 + r4; // 26 + r5 = r5 * r1; // 27 + r6 = mulhi(r6, r7); // 28 + r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29 + r6 = rotr_var(r6, r7); // 30 + r3 = r3 ^ dataset[r1 & MASK]; // 31 + { uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 + r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33 + { uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 + r0 = r0 * r3; // 35 + r2 = r2 ^ r5; // 36 + { uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 + r1 = r3 * r5 + r1; // 38 + r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39 + r2 = rotr_var(r2, r5); // 40 + r3 = r3 * r2; // 41 + r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42 + r3 = r3 ^ r4; // 43 + { uint s_ = r5 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 + r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45 + r7 = r7 ^ r1; // 46 + r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47 + r7 = mulhi(r7, r5); // 48 + r0 = r0 ^ dataset[r2 & MASK]; // 49 + r2 = r2 - r6; // 50 + r7 = r7 - r5; // 51 + r2 = r2 ^ r3; // 52 + r7 = r7 - r0; // 53 + r3 = r5 * r0 + r3; // 54 + r7 = r7 ^ r5; // 55 + r2 = r2 ^ dataset[r7 & MASK]; // 56 + r5 = r5 - r6; // 57 + r1 = r1 ^ dataset[r3 & MASK]; // 58 + { uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 + r4 = r4 - r6; // 60 + r1 = r1 * r2; // 61 + r3 = r6 * r0 + r3; // 62 + r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63 + } + uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u); + uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u); + out[gid] = ((ulong)hi << 32) | (ulong)lo; + } +} diff --git a/proto-cuda/packs-readwidth/scr8k32/vectors.h b/proto-cuda/packs-readwidth/scr8k32/vectors.h new file mode 100644 index 000000000..3d095f577 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k32/vectors.h @@ -0,0 +1,57 @@ +// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand. +// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset +#pragma once +#ifdef __cplusplus +#include +#else +#include +#endif + +#define IGNEUM_VEC_WARPS 3 +static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u }; +static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = { + { // base nonce 0 + 0x82097370c10ee4eaull, 0xa0f16965a422e758ull, 0x2b7513e2593b170cull, 0x8fbb69f8974c4533ull, 0x500c384970d19f32ull, 0x72fb96e1820fa2b8ull, 0xca1521df6a934a65ull, 0x7407397e10548cbcull, + 0xa619f127d1aa896bull, 0x36ce1104c1517d86ull, 0x6078ddd46d6d26b2ull, 0x2d22ded89010fbd3ull, 0xb030b743355bf6f9ull, 0x8280c8c1f4946ef7ull, 0xbfdae5712c7b9ad4ull, 0x50e9136a89e2e046ull, + 0xfe0230d874a94f3aull, 0x8b6dbb8ff0546c32ull, 0x41438f3ab2d35cbcull, 0x6faf2dd5c5997f9cull, 0x52c3aae40019a0abull, 0xc455bf9926661472ull, 0x4eaf9e34af6c7c22ull, 0x9dcd05c3a3991115ull, + 0x3a4f0ed406a66796ull, 0x7615703c08eabd8aull, 0x6007bcf34b8c6341ull, 0x1fb11900437441aeull, 0x540cff895991c6acull, 0x367a10121393f054ull, 0xf0dd7aa43fbdc2f4ull, 0x53d26f9c120f8a64ull + }, + { // base nonce 4096 + 0xc04adc4c89d1593bull, 0xc938ee393ee383dcull, 0xaef535444b6968fbull, 0x83581d97cebb4b30ull, 0x365315d0762bf586ull, 0xc26dfc2c40131920ull, 0x2f560c6fa351ead5ull, 0xc235996f6596d536ull, + 0x1adf3384c9f23712ull, 0x0c87a8ad0dc872c3ull, 0x82471bfd2a3182d4ull, 0xe8828b0fb7560877ull, 0x3fd9044dd153dd7full, 0x3a91a9618ce2c525ull, 0xf8c503abac8481f9ull, 0x56b0989f6014f8dbull, + 0xa1107a2732c588baull, 0xfc47a39e530d7efaull, 0x9237f7fc01727703ull, 0x01b5cbb25e6629c1ull, 0xae8e66fcec949f3eull, 0xb31655fbd75d3dbdull, 0x5a7750b585b3d0f1ull, 0x72f0484703cf30a0ull, + 0x4cea6de3f76e970bull, 0x64fd914fb1fb5096ull, 0xbdcca26f07e7182cull, 0xc86ba79905164a5bull, 0x521b7d3ac36f06cbull, 0x6b8618a80cef2e71ull, 0x13df43bef1f7852full, 0x202e92ce7b597a33ull + }, + { // base nonce 1000000 + 0x4db5bf37f24811f2ull, 0x5e93450594a45e5full, 0xebe57bb921f163aaull, 0x0457e6f5ac702b1eull, 0x442d8926fceec0d4ull, 0x417256d19cce43d3ull, 0x3d61563a8303daceull, 0x4a43c5efac6f6d47ull, + 0x9ef8a8d0a98f5c7eull, 0xbee88175926bb251ull, 0xb3b8e73f1e427be1ull, 0x515405b57446beecull, 0xba4f1765e616bf9aull, 0x148e5d9895c48299ull, 0x303fed05bdcdd7d1ull, 0x44d07bf31dba804full, + 0x8616fd225f3851beull, 0x426a7a80ac1b462full, 0x6e0163361c5ec30full, 0x065b3666feb8d0e5ull, 0xf9bc697886c9983eull, 0x90bc61f358b511f4ull, 0xfa47afe197811c66ull, 0x12a39e4e67aa2e97ull, + 0xde01f49ccab26a42ull, 0x2c6e837e74897413ull, 0x62d22c8acdeb1d09ull, 0x7fa8da035f65bb0bull, 0xdc4ce47ce0d48b6bull, 0x6199717653754041ull, 0x3a5e113c0d160d86ull, 0x718b3357f2391b60ull + } +}; + +// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455). +static const uint32_t IGNEUM_DS_HEAD[16] = { + 0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu, + 0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du +}; +static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u; +static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u; +// 64 sampled dataset words (index, value) computed on the Mac. +#define IGNEUM_DS_SAMPLES 64 +static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = { + 59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u +}; +static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = { + 0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u +}; +// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words. +static const uint32_t IGNEUM_CACHE_HEAD[16] = { + 0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u, + 0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u +}; +static const uint32_t IGNEUM_CACHE_LAST[16] = { + 0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du, + 0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu +}; +static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull; diff --git a/proto-cuda/packs-readwidth/scr8k32/vectors.json b/proto-cuda/packs-readwidth/scr8k32/vectors.json new file mode 100644 index 000000000..ce548d538 --- /dev/null +++ b/proto-cuda/packs-readwidth/scr8k32/vectors.json @@ -0,0 +1,36 @@ +{ + "seed": "igneum-genesis", + "day": "2026-10-03", + "dataset_mode": "memory-hard", + "dataset_log2_words": 28, + "mask": "0x0fffffff", + "lanes": 32, + "source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset", + "warps": [ + {"base_nonce": 0, "expected": [ + "0x82097370c10ee4ea", "0xa0f16965a422e758", "0x2b7513e2593b170c", "0x8fbb69f8974c4533", "0x500c384970d19f32", "0x72fb96e1820fa2b8", "0xca1521df6a934a65", "0x7407397e10548cbc", + "0xa619f127d1aa896b", "0x36ce1104c1517d86", "0x6078ddd46d6d26b2", "0x2d22ded89010fbd3", "0xb030b743355bf6f9", "0x8280c8c1f4946ef7", "0xbfdae5712c7b9ad4", "0x50e9136a89e2e046", + "0xfe0230d874a94f3a", "0x8b6dbb8ff0546c32", "0x41438f3ab2d35cbc", "0x6faf2dd5c5997f9c", "0x52c3aae40019a0ab", "0xc455bf9926661472", "0x4eaf9e34af6c7c22", "0x9dcd05c3a3991115", + "0x3a4f0ed406a66796", "0x7615703c08eabd8a", "0x6007bcf34b8c6341", "0x1fb11900437441ae", "0x540cff895991c6ac", "0x367a10121393f054", "0xf0dd7aa43fbdc2f4", "0x53d26f9c120f8a64" + ]}, + {"base_nonce": 4096, "expected": [ + "0xc04adc4c89d1593b", "0xc938ee393ee383dc", "0xaef535444b6968fb", "0x83581d97cebb4b30", "0x365315d0762bf586", "0xc26dfc2c40131920", "0x2f560c6fa351ead5", "0xc235996f6596d536", + "0x1adf3384c9f23712", "0x0c87a8ad0dc872c3", "0x82471bfd2a3182d4", "0xe8828b0fb7560877", "0x3fd9044dd153dd7f", "0x3a91a9618ce2c525", "0xf8c503abac8481f9", "0x56b0989f6014f8db", + "0xa1107a2732c588ba", "0xfc47a39e530d7efa", "0x9237f7fc01727703", "0x01b5cbb25e6629c1", "0xae8e66fcec949f3e", "0xb31655fbd75d3dbd", "0x5a7750b585b3d0f1", "0x72f0484703cf30a0", + "0x4cea6de3f76e970b", "0x64fd914fb1fb5096", "0xbdcca26f07e7182c", "0xc86ba79905164a5b", "0x521b7d3ac36f06cb", "0x6b8618a80cef2e71", "0x13df43bef1f7852f", "0x202e92ce7b597a33" + ]}, + {"base_nonce": 1000000, "expected": [ + "0x4db5bf37f24811f2", "0x5e93450594a45e5f", "0xebe57bb921f163aa", "0x0457e6f5ac702b1e", "0x442d8926fceec0d4", "0x417256d19cce43d3", "0x3d61563a8303dace", "0x4a43c5efac6f6d47", + "0x9ef8a8d0a98f5c7e", "0xbee88175926bb251", "0xb3b8e73f1e427be1", "0x515405b57446beec", "0xba4f1765e616bf9a", "0x148e5d9895c48299", "0x303fed05bdcdd7d1", "0x44d07bf31dba804f", + "0x8616fd225f3851be", "0x426a7a80ac1b462f", "0x6e0163361c5ec30f", "0x065b3666feb8d0e5", "0xf9bc697886c9983e", "0x90bc61f358b511f4", "0xfa47afe197811c66", "0x12a39e4e67aa2e97", + "0xde01f49ccab26a42", "0x2c6e837e74897413", "0x62d22c8acdeb1d09", "0x7fa8da035f65bb0b", "0xdc4ce47ce0d48b6b", "0x6199717653754041", "0x3a5e113c0d160d86", "0x718b3357f2391b60" + ]} + ], + "dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"], + "dataset_last_index": 268435455, + "dataset_last": "0xa33ada72", + "dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}], + "cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"], + "cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"], + "cache_fnv1a64": "0x48c4f5bf24166b2e" +} diff --git a/proto-metal/packbench.swift b/proto-metal/packbench.swift new file mode 100644 index 000000000..75b6fc4f3 --- /dev/null +++ b/proto-metal/packbench.swift @@ -0,0 +1,209 @@ +// packbench: runs a program pack (igneum-pow export) on the Mac's Metal GPU from its files alone: memhard.metal (cache +// fill, dataset build), program.metal (igneum_hash), program.h (constants), vectors.json (the CPU reference's vectors). +// Read-width experiment, 5 October 2026 (docs/plans/read-width.md): the Swift bench generates its own programs and does +// not know the experiment's load classes; this harness runs whatever text the Rust emitter wrote, so Metal is checked +// against the Rust CPU reference and timed without a Swift mirror of the generator. One file, no packages. +// +// swiftc -O -target arm64-apple-macos11 -o packbench packbench.swift -framework Metal +// ./packbench --pack [--batches 5] [--batch-log2 24] [--group 256] [--warps 2048] +// +// Prints one RESULT line per run: vectors, cache and dataset checks, the batch fingerprint (FNV-1a 64 over the 2^B +// outputs at base nonce 0) and MH/s by wall and by GPU time. Variant 5 packs (IGNEUM_PERSISTENT_WARPS) are launched +// as --warps persistent warps with a 1 MiB scratch each; the batch is rounded to a multiple of 32 x warps. +import Foundation +import Metal + +func nowMs() -> Double { return Double(DispatchTime.now().uptimeNanoseconds) / 1e6 } +func fail(_ m: String) -> Never { print("FAIL: \(m)"); exit(1) } + +struct Opts { var pack = ""; var batches = 5; var batchLog2 = 24; var group = 256; var warps = 2048 } +var opts = Opts() +var args = Array(CommandLine.arguments.dropFirst()) +while !args.isEmpty { + let a = args.removeFirst() + func next() -> String { if args.isEmpty { fail("missing value for \(a)") }; return args.removeFirst() } + switch a { + case "--pack": opts.pack = next() + case "--batches": opts.batches = Int(next())! + case "--batch-log2": opts.batchLog2 = Int(next())! + case "--group": opts.group = Int(next())! + case "--warps": opts.warps = Int(next())! + default: fail("unknown argument \(a)") + } +} +if opts.pack.isEmpty { fail("--pack is required") } + +func readText(_ name: String) -> String { + guard let s = try? String(contentsOfFile: opts.pack + "/" + name, encoding: .utf8) else { fail("cannot read \(opts.pack)/\(name)") } + return s +} +let programH = readText("program.h") +func defineU32(_ name: String) -> UInt32? { + let pat = "#define \(name) ([0-9a-fA-Fx]+)" + guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil } + let v = String(programH[Range(m.range(at: 1), in: programH)!]).replacingOccurrences(of: "u", with: "") + if v.hasPrefix("0x") { return UInt32(v.dropFirst(2), radix: 16) } + return UInt32(v) +} +func defineStr(_ name: String) -> String? { + let pat = "#define \(name) \"([^\"]*)\"" + guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil } + return String(programH[Range(m.range(at: 1), in: programH)!]) +} +let datasetLog2 = Int(defineU32("IGNEUM_DATASET_LOG2") ?? 28) +let datasetMode = defineU32("IGNEUM_DATASET_MODE") ?? 1 +if datasetMode != 1 { fail("packbench runs memory-hard packs only") } +let cacheLog2 = Int(defineU32("IGNEUM_CACHE_LOG2_WORDS") ?? 26) +let cacheSegments = Int(defineU32("IGNEUM_CACHE_SEGMENTS") ?? 65536) +let loadsPerHash = Int(defineU32("IGNEUM_LOADS_PER_HASH") ?? 128) +let bytesPerHash = Int(defineU32("IGNEUM_BYTES_PER_HASH") ?? UInt32(loadsPerHash * 4)) +let scratchOps = Int(defineU32("IGNEUM_SCRATCH_OPS") ?? 0) +let persistent = (defineU32("IGNEUM_PERSISTENT_WARPS") ?? 0) == 1 +let scratchWordsPerLane = Int(defineU32("IGNEUM_SCRATCH_WORDS_PER_LANE") ?? 8192) +let className = defineStr("IGNEUM_LOAD_CLASS") ?? "v2" +let seedString = defineStr("IGNEUM_SEED_STRING") ?? "?" +let programId = defineStr("IGNEUM_PROGRAM_ID") ?? "" + +// vectors.json: bases, expected outputs, dataset head / last, cache fingerprint +let vj = try! JSONSerialization.jsonObject(with: Data(contentsOf: URL(fileURLWithPath: opts.pack + "/vectors.json"))) as! [String: Any] +func hex64(_ s: String) -> UInt64 { return UInt64(s.dropFirst(2), radix: 16)! } +func hex32(_ s: String) -> UInt32 { return UInt32(s.dropFirst(2), radix: 16)! } +let warpsJ = vj["warps"] as! [[String: Any]] +let vecBases = warpsJ.map { UInt32(($0["base_nonce"] as! NSNumber).uint64Value) } +let vecOuts = warpsJ.map { ($0["expected"] as! [String]).map(hex64) } +let dsHead = (vj["dataset_head"] as! [String]).map(hex32) +let dsLastIndex = UInt32((vj["dataset_last_index"] as! NSNumber).uint64Value) +let dsLast = hex32(vj["dataset_last"] as! String) +let cacheFnvWant = hex64(vj["cache_fnv1a64"] as! String) + +guard let device = MTLCreateSystemDefaultDevice(), let queue = device.makeCommandQueue() else { fail("no Metal device") } +let words = 1 << datasetLog2 +let mask = UInt32(words - 1) +let cacheWords = 1 << cacheLog2 + +func compile(_ file: String) -> MTLLibrary { + do { return try device.makeLibrary(source: readText(file), options: MTLCompileOptions()) } catch { fail("Metal compile of \(file): \(error)") } +} +let t0 = nowMs() +let mhLib = compile("memhard.metal") +let progLib = compile("program.metal") +guard let fillFn = mhLib.makeFunction(name: "igneum_cache_fill"), let buildFn = mhLib.makeFunction(name: "igneum_build"), let hashFn = progLib.makeFunction(name: "igneum_hash") else { fail("kernel functions missing") } +let fillPipe = try! device.makeComputePipelineState(function: fillFn) +let buildPipe = try! device.makeComputePipelineState(function: buildFn) +let hashPipe = try! device.makeComputePipelineState(function: hashFn) +let compileMs = nowMs() - t0 +if hashPipe.threadExecutionWidth != 32 { print("WARNING: threadExecutionWidth \(hashPipe.threadExecutionWidth), not 32") } + +guard let cache = device.makeBuffer(length: cacheWords * 4, options: .storageModePrivate) else { fail("cache alloc") } +guard let dataset = device.makeBuffer(length: words * 4, options: .storageModePrivate) else { fail("dataset alloc") } + +func run(_ body: (MTLComputeCommandEncoder) -> Void) -> (Double, Double) { + let cb = queue.makeCommandBuffer()! + let enc = cb.makeComputeCommandEncoder()! + body(enc) + enc.endEncoding() + let w0 = nowMs() + cb.commit(); cb.waitUntilCompleted() + if let e = cb.error { fail("command buffer: \(e)") } + return (nowMs() - w0, (cb.gpuEndTime - cb.gpuStartTime) * 1000) +} +let (cacheWall, cacheGpu) = run { enc in + enc.setComputePipelineState(fillPipe); enc.setBuffer(cache, offset: 0, index: 0) + enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1)) +} +let items = words / 16 +let (buildWall, buildGpu) = run { enc in + enc.setComputePipelineState(buildPipe); enc.setBuffer(cache, offset: 0, index: 0); enc.setBuffer(dataset, offset: 0, index: 1) + enc.dispatchThreadgroups(MTLSize(width: items / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1)) +} +// cache fingerprint and dataset head/last through a blit to shared memory +func blit(_ src: MTLBuffer, _ offset: Int, _ n: Int) -> MTLBuffer { + let dst = device.makeBuffer(length: n, options: .storageModeShared)! + let cb = queue.makeCommandBuffer()!; let b = cb.makeBlitCommandEncoder()! + b.copy(from: src, sourceOffset: offset, to: dst, destinationOffset: 0, size: n); b.endEncoding(); cb.commit(); cb.waitUntilCompleted() + return dst +} +func fnv1a64(_ p: UnsafeRawPointer, _ n: Int) -> UInt64 { + var h: UInt64 = 0xcbf29ce484222325 + let b = p.bindMemory(to: UInt8.self, capacity: n) + for i in 0..> 20) MiB") } + scratch = s +} +// One hash launch: `nonces` outputs from `base`. Persistent: warpsN warps loop over nonces / 32 units. +func encodeHash(_ enc: MTLComputeCommandEncoder, out: MTLBuffer, base: UInt32, nonces: Int, group: Int) { + enc.setComputePipelineState(hashPipe) + enc.setBuffer(dataset, offset: 0, index: 0) + enc.setBuffer(out, offset: 0, index: 1) + var b = base; enc.setBytes(&b, length: 4, index: 2) + if persistent { + let units = nonces / 32 + let nw = min(warpsN, units) + if units % nw != 0 { fail("nonces \(nonces) is not a multiple of 32 x \(nw) warps") } + enc.setBuffer(scratch!, offset: 0, index: 3) + var g = UInt32(units); enc.setBytes(&g, length: 4, index: 4) + var s = salt; enc.setBytes(&s, length: 4, index: 5) + salt = salt &+ UInt32(units) + let threads = nw * 32 + let tg = min(group, threads) + enc.dispatchThreadgroups(MTLSize(width: threads / tg, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1)) + } else { + let tg = min(group, nonces) + enc.dispatchThreadgroups(MTLSize(width: nonces / tg, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1)) + } +} +// Vectors, standalone (one unit per launch) +var vecPass = 0 +let vecOut = device.makeBuffer(length: 32 * 8, options: .storageModeShared)! +for (i, base) in vecBases.enumerated() { + _ = run { enc in encodeHash(enc, out: vecOut, base: base, nonces: 32, group: 32) } + let p = vecOut.contents().bindMemory(to: UInt64.self, capacity: 32) + var ok = true + for l in 0..<32 where p[l] != vecOuts[i][l] { ok = false; print("vector warp base \(base) lane \(l): GPU \(String(format: "%016llx", p[l])) expected \(String(format: "%016llx", vecOuts[i][l]))"); break } + if ok { vecPass += 1 } +} +// Batch at base 0: fingerprint and the vectors inside the batch +var nonces = 1 << opts.batchLog2 +if persistent { let unit = 32 * min(warpsN, nonces / 32); nonces = (nonces / unit) * unit } +let out = device.makeBuffer(length: nonces * 8, options: .storageModeShared)! +let (warmWall, warmGpu) = run { enc in encodeHash(enc, out: out, base: 0, nonces: nonces, group: opts.group) } +let outPtr = out.contents().bindMemory(to: UInt64.self, capacity: nonces) +var batchVecPass = 0, batchVecN = 0 +for (i, base) in vecBases.enumerated() where Int(base) + 32 <= nonces { + batchVecN += 1 + if (0..<32).allSatisfy({ outPtr[Int(base) + $0] == vecOuts[i][$0] }) { batchVecPass += 1 } +} +let fingerprint = fnv1a64(out.contents(), nonces * 8) +// Timed batches +var wallSum = 0.0, gpuSum = 0.0 +for b in 0.. out(nonces); +#ifdef IGNEUM_PERSISTENT_WARPS + // Variant 5: EMU_WARPS persistent warps (one per work-group of 32), each with IGNEUM_SCRATCH_WORDS_PER_LANE x 32 words. + const unsigned emuWarps = 8; + std::vector scratchArena((size_t)emuWarps * 32u * IGNEUM_SCRATCH_WORDS_PER_LANE, 0u); + uint salt = 1u; + if (IGNEUM_GROUP != 32) { std::printf("variant 5 needs IGNEUM_GROUP 32 (one warp per work-group)\n"); return 2; } + std::printf("variant 5: %u persistent warps, scratch arena %u MiB, lazy tagged fill\n", emuWarps, (unsigned)((scratchArena.size() * 4u) >> 20)); + auto launchHash = [&](size_t units, uint base) { + size_t nw = units < emuWarps ? units : emuWarps; + emu_launch(igneum_hash, nw * 32u, 32u, (const uint*)ds.data(), out.data(), base, mask, scratchArena.data(), (uint)units, salt); + salt += (uint)units; + }; +#else + auto launchHash = [&](size_t nonceCount, uint base) { + emu_launch(igneum_hash, nonceCount, (unsigned)IGNEUM_GROUP, (const uint*)ds.data(), out.data(), base, mask); + }; +#endif for (int w = 0; w < IGNEUM_VEC_WARPS; ++w) { - emu_launch(igneum_hash, (size_t)IGNEUM_GROUP, (unsigned)IGNEUM_GROUP, (const uint*)ds.data(), out.data(), (uint)IGNEUM_VEC_BASE[w], mask); +#ifdef IGNEUM_PERSISTENT_WARPS + launchHash(1, (uint)IGNEUM_VEC_BASE[w]); +#else + launchHash((size_t)IGNEUM_GROUP, (uint)IGNEUM_VEC_BASE[w]); +#endif char how[96]; std::snprintf(how, sizeof(how), "standalone, work-group %d, sub-group width %u", IGNEUM_GROUP, gSubGroupWidth); overall = compareWarp((const uint64_t*)out.data(), IGNEUM_VEC_OUT[w], IGNEUM_VEC_BASE[w], how) && overall; } // In batch: every vector warp that fits in 2^batchLog2 nonces. - emu_launch(igneum_hash, (size_t)nonces, (unsigned)IGNEUM_GROUP, (const uint*)ds.data(), out.data(), 0u, mask); +#ifdef IGNEUM_PERSISTENT_WARPS + launchHash(nonces / 32u, 0u); +#else + launchHash((size_t)nonces, 0u); +#endif for (int w = 0; w < IGNEUM_VEC_WARPS; ++w) { if ((uint64_t)IGNEUM_VEC_BASE[w] + 32ull > nonces) { std::printf("verify warp base %u in batch: skipped (batch has %u nonces)\n", IGNEUM_VEC_BASE[w], nonces); continue; } char how[96]; diff --git a/proto-opencl/host.c b/proto-opencl/host.c index 31be99b18..bd04cfe56 100644 --- a/proto-opencl/host.c +++ b/proto-opencl/host.c @@ -207,6 +207,9 @@ typedef struct { int memprobe; // --memprobe: dependent-load latency and throughput, independent-load throughput and an ALU // chain on the chosen device, no pack needed (5 October 2026, the 9070 XT on the eGPU) int probeMib; // --probe-mib N: --memprobe at that one buffer size only (default 0 = 4, 64 and 1024 MiB) + int benchPack; // --bench-pack: with --pack D, build and self-test the pack at run time (as --serve does) and time + // igneum_hash_bound with the pack's seed words as init words; read-width experiment, 5 October 2026 + int warps; // --warps N: persistent warps for a variant-5 pack (IGNEUM_PERSISTENT_WARPS); 0 = 2048 } Options; static int packMib(void) { return (int)(((1ull << IGNEUM_DATASET_LOG2) * 4ull) >> 20); } @@ -236,6 +239,8 @@ static void usage(void) { " pack this exe was built against; it is self-tested against its vectors.h first (the one-click worker)\n" " --readback M with --serve: select (default) reads back only the hits and 34 sentinel words of each dispatch through a\n" " GPU-side pass; full reads back every output (8 bytes per nonce). IGNEUM_READBACK=full does the same.\n" + " --bench-pack with --pack D: read the pack at run time, build and self-test it, time its bound kernel (one exe, any pack)\n" + " --warps N persistent warps for a variant-5 pack (a 1 MiB scratch each; default 2048; the batch rounds to 32 x N)\n" " --memprobe no pack: dependent random loads (latency and throughput against lanes in flight), independent random\n" " loads and an ALU chain on the chosen device, at 4, 64 and 1024 MiB (--probe-mib N for one size)\n", packMib(), IGNEUM_KERNEL_PATH); } @@ -248,7 +253,7 @@ static Options parseArgs(int argc, char** argv) { int i; o.datasetMib = 1024; o.batchLog2 = 24; o.batches = 5; o.groupWarps = 1; o.sweep = 0; o.device = -1; o.exchange = 0; o.list = 0; o.timeWall = -1; o.kernelPath = IGNEUM_KERNEL_PATH; o.extraOpts = ""; o.serve = 0; o.noPrepare = 0; o.kernelGiven = 0; o.vendor = NULL; o.packDir = NULL; - o.readback = (getenv("IGNEUM_READBACK") && strcmp(getenv("IGNEUM_READBACK"), "full") == 0) ? 1 : 0; o.memprobe = 0; o.probeMib = 0; + o.readback = (getenv("IGNEUM_READBACK") && strcmp(getenv("IGNEUM_READBACK"), "full") == 0) ? 1 : 0; o.memprobe = 0; o.probeMib = 0; o.benchPack = 0; o.warps = 0; for (i = 1; i < argc; ++i) { const char* a = argv[i]; int needs = (strcmp(a, "--dataset-mib") == 0 || strcmp(a, "--batch-log2") == 0 || strcmp(a, "--batches") == 0 || @@ -274,6 +279,8 @@ static Options parseArgs(int argc, char** argv) { else { printf("--readback must be select or full\n"); exit(2); } } else if (strcmp(a, "--memprobe") == 0) o.memprobe = 1; + else if (strcmp(a, "--bench-pack") == 0) o.benchPack = 1; + else if (strcmp(a, "--warps") == 0) { if (i + 1 >= argc) { usage(); exit(2); } o.warps = atoi(argv[++i]); } else if (strcmp(a, "--probe-mib") == 0) { if (i + 1 >= argc) { usage(); exit(2); } o.probeMib = atoi(argv[++i]); } else if (strcmp(a, "--build-opts") == 0) o.extraOpts = argv[++i]; else if (strcmp(a, "--time") == 0) { @@ -1013,6 +1020,19 @@ static int unhexBuf(const char* s, uint8_t* out, size_t cap, size_t* len) { * from the compiled-in program.h, so one prebuilt exe serves every pack. The compiled-in values are the defaults. */ static PfPack gPack; static int gGeneric = 0; +/* Variant 5 of the read-width experiment (5 October 2026): the scratch arena of a persistent-warp pack, its warp count + * and the running tag salt; set by --bench-pack before the self-test. Serve mode does not support these packs. */ +static cl_mem gScratch = NULL; +static cl_uint gScratchWarps = 0; +static cl_uint gSalt = 1; +/* Sets the three extra arguments of a variant-5 kernel (after the five of igneum_hash_bound) for `units` units. */ +static cl_int setScratchArgs(cl_kernel k, cl_uint firstArg, cl_uint units) { + cl_int e = clSetKernelArg(k, firstArg, sizeof(cl_mem), &gScratch); + if (e == CL_SUCCESS) e = clSetKernelArg(k, firstArg + 1, sizeof(cl_uint), &units); + if (e == CL_SUCCESS) e = clSetKernelArg(k, firstArg + 2, sizeof(cl_uint), &gSalt); + gSalt += units; + return e; +} static uint32_t gServeWords = 1u << IGNEUM_DATASET_LOG2; #if IGNEUM_DATASET_MODE == 1 static uint32_t gServeCacheWords = 1u << IGNEUM_CACHE_LOG2_WORDS; @@ -1102,6 +1122,10 @@ static int pairSelfTest(Device* dv, const DeviceInfo* di, cl_command_queue q, Se out = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, g * sizeof(uint64_t), NULL, &e); if (e == CL_SUCCESS) { ++gMemCreated; init = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY | CL_MEM_COPY_HOST_PTR, 32, pk.seedw, &e); } if (e == CL_SUCCESS) ++gMemCreated; + if (pk.persistent) { + if (!gScratch) { snprintf(err, errCap, "self-test: a variant-5 pack (persistent warps) needs --bench-pack (serve mode does not carry a scratch)"); return 0; } + g = 32; local = 32; /* one persistent warp runs the one unit */ + } for (w = 0; w < pk.vecWarps && e == CL_SUCCESS; ++w) { cl_uint base = pk.vecBase[w], mask = words - 1u; e = clSetKernelArg(p->kHashBound, 0, sizeof(cl_mem), &p->ds); @@ -1109,6 +1133,7 @@ static int pairSelfTest(Device* dv, const DeviceInfo* di, cl_command_queue q, Se if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 2, sizeof(cl_uint), &base); if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 3, sizeof(cl_uint), &mask); if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 4, sizeof(cl_mem), &init); + if (e == CL_SUCCESS && pk.persistent) e = setScratchArgs(p->kHashBound, 5, 1u); if (e == CL_SUCCESS) e = clEnqueueNDRangeKernel(q, p->kHashBound, 1, NULL, &g, &local, 0, NULL, NULL); if (e == CL_SUCCESS) e = clEnqueueReadBuffer(q, out, CL_TRUE, 0, 32 * sizeof(uint64_t), &vec[w * 32], 0, NULL, NULL); } @@ -1209,6 +1234,92 @@ static int startPrepareThread(PrepareTask* t) { pthread_t th; if (pthread_create #endif #endif +/* --bench-pack (read-width experiment, 5 October 2026): the pack in --pack is built and self-tested exactly as the + * first pair of --serve (pairBuffers: cache, dataset, cache FNV, dataset words, the vector warps through + * igneum_hash_bound with the pack's seed words), then the bound kernel is timed over --batches dispatches of + * 2^--batch-log2 nonces with device event time, and the 2^B outputs at base nonce 0 are fingerprinted (FNV-1a 64) so + * the same pack can be compared bit for bit across vendors. A variant-5 pack is launched as --warps persistent warps + * with a 1 MiB scratch each. One line per run starts with RESULT. */ +static int runBenchPack(Device* dv, const DeviceInfo* di, const Options* o) { +#if IGNEUM_DATASET_MODE != 1 + (void)dv; (void)di; (void)o; + printf("FAIL: --bench-pack needs a memory-hard placeholder pack\n"); + return 2; +#else + const uint32_t words = gServeWords, mask = words - 1u; + uint32_t nonces = 1u << o->batchLog2; + size_t groupSize = 32 * (size_t)o->groupWarps, g; + cl_int err = 0; + cl_mem dOut, dInit; + uint64_t* hOut; + ServePair* cur; + char perr[512], devName[256]; + double t0 = wallMs(), sum = 0, warmMs; + uint64_t fp; + int b, k; + cl_uint warps = (cl_uint)(o->warps > 0 ? o->warps : 2048), units = nonces / 32u; + if (!dv->kHashBound) { printf("FAIL: the kernel source has no igneum_hash_bound\n"); return 2; } + if (gPack.persistent) { + size_t arena; + if (o->groupWarps != 1) { printf("FAIL: a variant-5 pack needs --group-warps 1 (one warp per work-group: the loop trip count must be uniform)\n"); return 2; } + while (warps > 1 && units % warps != 0) warps >>= 1; + arena = (size_t)warps * 32u * (size_t)gPack.scratchWordsPerLane * 4u; + if ((uint64_t)arena > di->maxAlloc) { printf("FAIL: scratch arena %llu MiB exceeds the device's max alloc %llu MiB; lower --warps\n", (unsigned long long)(arena >> 20), (unsigned long long)(di->maxAlloc >> 20)); return 2; } + gScratch = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, arena, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer scratch"); + gScratchWarps = warps; + printf("variant 5: %u persistent warps (%u x 32 work-items, work-group 32), scratch arena %llu MiB, %u units per dispatch, lazy tagged fill\n", warps, warps, (unsigned long long)(arena >> 20), units); + } + cur = (ServePair*)calloc(1, sizeof(ServePair)); + cur->kHashBound = dv->kHashBound; cur->kCacheFill = dv->kCacheFill; cur->kBuild = dv->kBuild; cur->prog = dv->prog; + dv->kHashBound = dv->kCacheFill = dv->kBuild = NULL; dv->prog = NULL; + memcpy(cur->sw, gPack.seedw, 32); memcpy(cur->kw, gPack.keyw, 32); + if (!pairBuffers(dv, di, dv->q, cur, words, gServeCacheWords, gServeSegments, o->packDir, perr, sizeof(perr))) { printf("FAIL: pack %s: %s\n", o->packDir, perr); return 1; } + printf("pack %s: cache %.0f dataset %.0f check %.0f ms (%.0f ms in all); %s\n", o->packDir, cur->cacheMs, cur->datasetMs, cur->checkMs, wallMs() - t0, cur->check); + printKernelInfo(di, cur->kHashBound, "igneum_hash_bound", (int)groupSize, ""); + dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)nonces * sizeof(uint64_t), NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer out"); + dInit = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY | CL_MEM_COPY_HOST_PTR, 32, cur->sw, &err); CL_CHECK_ERR(err, "clCreateBuffer init words"); + hOut = (uint64_t*)malloc((size_t)nonces * sizeof(uint64_t)); + strncpy(devName, di->name, 255); devName[255] = 0; + for (k = 0; devName[k]; ++k) if (devName[k] == ' ') devName[k] = '_'; + g = gPack.persistent ? (size_t)warps * 32u : (size_t)nonces; + for (b = -1; b < o->batches; ++b) { + cl_uint base = (cl_uint)((uint32_t)(b + 1) * nonces); + cl_event ev = NULL; + double ms; + CL_CHECK(clSetKernelArg(cur->kHashBound, 0, sizeof(cl_mem), &cur->ds)); + CL_CHECK(clSetKernelArg(cur->kHashBound, 1, sizeof(cl_mem), &dOut)); + CL_CHECK(clSetKernelArg(cur->kHashBound, 2, sizeof(cl_uint), &base)); + CL_CHECK(clSetKernelArg(cur->kHashBound, 3, sizeof(cl_uint), &mask)); + CL_CHECK(clSetKernelArg(cur->kHashBound, 4, sizeof(cl_mem), &dInit)); + if (gPack.persistent) CL_CHECK(setScratchArgs(cur->kHashBound, 5, units)); + { + double w0 = wallMs(); + CL_CHECK(clEnqueueNDRangeKernel(dv->q, cur->kHashBound, 1, NULL, &g, &groupSize, 0, NULL, &ev)); + CL_CHECK(clWaitForEvents(1, &ev)); + ms = o->timeWall ? wallMs() - w0 : eventMs(ev); + if (ms < 0) ms = wallMs() - w0; + clReleaseEvent(ev); + } + if (b < 0) { + warmMs = ms; + CL_CHECK(clEnqueueReadBuffer(dv->q, dOut, CL_TRUE, 0, (size_t)nonces * sizeof(uint64_t), hOut, 0, NULL, NULL)); + fp = pf_fnv1a64((const uint32_t*)hOut, (size_t)nonces * 8u); + } else sum += ms; + } + printf("warm-up dispatch (base 0): %.2f ms; %d timed dispatches of 2^%d nonces: mean %.2f ms\n", warmMs, o->batches, o->batchLog2, sum / o->batches); + printf("RESULT pack=%s class=%s device=%s platform=%s group=%d warps=%u arena_mib=%llu nonces=%u batches=%d check=%s fingerprint=%016llx mhs=%.3f loads=%u bytes=%u scratch_ops=%u time=%s\n", + o->packDir, gPack.loadClass, devName, strcmp(di->platformName, "Apple") == 0 ? "Apple" : "other", (int)groupSize, gPack.persistent ? warps : 0u, + gPack.persistent ? (unsigned long long)(((size_t)warps * 32u * gPack.scratchWordsPerLane * 4u) >> 20) : 0ull, nonces, o->batches, + cur->checked ? "PASS" : "skipped", (unsigned long long)fp, (double)nonces * (double)o->batches / (sum / 1000.0) / 1e6, + gPack.loadsPerHash, gPack.bytesPerHash, gPack.scratchOps * 8u, o->timeWall ? "wall" : "event"); + free(hOut); + clReleaseMemObject(dOut); clReleaseMemObject(dInit); + if (gScratch) clReleaseMemObject(gScratch); + releasePair(cur); + return 0; +#endif +} + static int runServe(Device* dv, const DeviceInfo* di, const Options* o) { #if IGNEUM_DATASET_MODE != 1 (void)dv; (void)di; (void)o; @@ -1612,6 +1723,11 @@ static const char* PROBE_SRC = " }\n" " out[g] = x0 ^ x1 ^ x2 ^ x3 ^ x4 ^ x5 ^ x6 ^ x7;\n" "}\n" + "__kernel void probe_line16(__global const uint4* ds, uint vecMask, uint steps, uint seed, __global uint* out) {\n" + " uint x = pm_mix((uint)get_global_id(0) ^ seed);\n" + " for (uint s = 0u; s < steps; ++s) { uint4 a = ds[x & vecMask]; x = (a.x ^ a.y ^ a.z ^ a.w) ^ (x * 0x9E3779B1u + s); }\n" + " out[get_global_id(0)] = x;\n" + "}\n" "__kernel void probe_line(__global const uint4* ds, uint lineMask, uint steps, uint seed, __global uint* out) {\n" " uint x = pm_mix((uint)get_global_id(0) ^ seed);\n" " for (uint s = 0u; s < steps; ++s) {\n" @@ -1656,7 +1772,7 @@ static double probeLaunch(Device* dv, const Options* o, cl_kernel k, size_t glob static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) { cl_int err = 0; cl_program prog; - cl_kernel kFill, kChase, kIndep, kAlu, kLine, kStream; + cl_kernel kFill, kChase, kIndep, kAlu, kLine, kStream, kLine16; size_t srcLen = strlen(PROBE_SRC); int sizes[3] = { 4, 64, 1024 }, nSizes = 3, si; size_t lanesList[8] = { 256, 1024, 1u << 12, 1u << 14, 1u << 16, 1u << 18, 1u << 20, 1u << 22 }; @@ -1682,6 +1798,7 @@ static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) { kIndep = clCreateKernel(prog, "probe_indep", &err); CL_CHECK_ERR(err, "probe_indep"); kAlu = clCreateKernel(prog, "probe_alu", &err); CL_CHECK_ERR(err, "probe_alu"); kLine = clCreateKernel(prog, "probe_line", &err); CL_CHECK_ERR(err, "probe_line"); + kLine16 = clCreateKernel(prog, "probe_line16", &err); CL_CHECK_ERR(err, "probe_line16"); kStream = clCreateKernel(prog, "probe_stream", &err); CL_CHECK_ERR(err, "probe_stream"); dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, maxLanes * 4u, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer probe out"); printf("memprobe on [%s] %s, driver %s, %u compute units, %u MHz, %s time\n", di->platformName, di->name, di->driver, di->computeUnits, di->clockMHz, o->timeWall ? "wall" : "device event"); @@ -1737,6 +1854,25 @@ static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) { fflush(stdout); } } + { + /* Random 16-byte reads (one uint4) in a dependent chain: the W = 16 width of the read-width experiment. */ + size_t local = di->maxWorkGroup < 256 ? di->maxWorkGroup : 256; + size_t lanes; + cl_uint vecMask = (words / 4u) - 1u; + for (lanes = 1u << 14; lanes <= maxLanes; lanes <<= 2) { + cl_uint seed = 0x2718281u; + double ms; + CL_CHECK(clSetKernelArg(kLine16, 0, sizeof(cl_mem), &dDs)); + CL_CHECK(clSetKernelArg(kLine16, 1, sizeof(cl_uint), &vecMask)); + CL_CHECK(clSetKernelArg(kLine16, 2, sizeof(cl_uint), &STEPS)); + CL_CHECK(clSetKernelArg(kLine16, 3, sizeof(cl_uint), &seed)); + CL_CHECK(clSetKernelArg(kLine16, 4, sizeof(cl_mem), &dOut)); + ms = probeLaunch(dv, o, kLine16, lanes, local, 3, 3, seed); + printf("| line 16 B | %d | %llu | %llu | %u | %.3f | %.3f G reads/s | %.1f GB/s in 16 B reads |\n", mib, (unsigned long long)local, (unsigned long long)lanes, STEPS, ms, + (double)lanes * (double)STEPS / (ms / 1000.0) / 1e9, (double)lanes * (double)STEPS * 16.0 / (ms / 1000.0) / 1e9); + fflush(stdout); + } + } { /* Random 64-byte lines (16 words, four uint4 loads) in a dependent chain: lines per second against the * 4-byte chase above says what one random 4-byte read costs the memory system. If the two rates are @@ -1792,7 +1928,7 @@ static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) { (double)lanes * (double)ALU_STEPS / (ms / 1000.0) / 1e9 / (double)(di->computeUnits ? di->computeUnits : 1)); } clReleaseMemObject(dOut); - clReleaseKernel(kFill); clReleaseKernel(kChase); clReleaseKernel(kIndep); clReleaseKernel(kAlu); clReleaseKernel(kLine); clReleaseKernel(kStream); + clReleaseKernel(kFill); clReleaseKernel(kChase); clReleaseKernel(kIndep); clReleaseKernel(kAlu); clReleaseKernel(kLine); clReleaseKernel(kStream); clReleaseKernel(kLine16); clReleaseProgram(prog); printf("memprobe: done\n"); return 0; @@ -1821,7 +1957,7 @@ int main(int argc, char** argv) { static char boundPath[1200]; char perr[512]; size_t n = strlen(o.packDir); - if (!o.serve) { printf("FAIL: --pack goes with --serve (the bench runs the compiled-in pack)\n"); return 2; } + if (!o.serve && !o.benchPack) { printf("FAIL: --pack goes with --serve or --bench-pack (the plain bench runs the compiled-in pack)\n"); return 2; } if (n > 1 && (o.packDir[n - 1] == '/' || o.packDir[n - 1] == '\\')) ((char*)o.packDir)[n - 1] = 0; if (!pf_load(o.packDir, &gPack, perr, sizeof(perr))) { printf("error 0 pack %s: %s\n", o.packDir, perr); fflush(stdout); return 2; } gGeneric = 1; @@ -1877,6 +2013,7 @@ int main(int argc, char** argv) { clReleaseContext(dv.ctx); return rc; } + if (o.benchPack && !o.packDir) { printf("FAIL: --bench-pack needs --pack \n"); return 2; } if (o.serve && !o.kernelGiven) { /* The bound kernel lives next to the compiled-in kernel.cl as kernel_bound.cl (packs from igneum-pow or igneum-miner export-pack). */ static char boundPath[1024]; @@ -1894,6 +2031,7 @@ int main(int argc, char** argv) { printf("build options: %s\n", dv.buildOptions); printf("exchange: %s\n", dv.exchangeNote); if (o.serve) return runServe(&dv, di, &o); + if (o.benchPack) return runBenchPack(&dv, di, &o); printKernelInfo(di, dv.kHash, "igneum_hash", dv.groupSize, ""); printf("program: %d instructions x %d iterations, loads/hash %d, op mix %s\n", IGNEUM_INSTR_COUNT, IGNEUM_ITERATIONS, IGNEUM_LOADS_PER_HASH, IGNEUM_OP_MIX);