read-width: scratch per warp is a class parameter (32 or 128 KiB, under the 6 GB working-set cap), distinct-address rule bounds dataset loads only; Metal pack harness; OpenCL --bench-pack, scratch args and 16-byte probe; packfile class fields; OpenCL emulator persistent launch

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-labs 2026-10-05 19:58:20 +00:00
parent 36d93e03c8
commit ea863d6d4b
92 changed files with 6031 additions and 367 deletions

View file

@ -14,7 +14,7 @@
//! costs about a millisecond on one core. The census (section 7.3) checked on 100,000 programs that the
//! closed-form verdict agrees with the memory-hard one on all but 39 threshold-edge cases.
use crate::generator::{Instr, Op, Program, INSTR_COUNT, ITERATIONS, LANES, SCRATCH_SLOT_MASK};
use crate::generator::{Instr, Op, Program, INSTR_COUNT, ITERATIONS, LANES};
use crate::seed::{fnv1a64, SplitMix64};
use crate::verify::{dataset_elem, fold_words, splitmix32, ScratchModel};
@ -33,8 +33,10 @@ pub const BIAS_TOLERANCE: u32 = 136;
/// Distinct addresses per lane per evaluation, summed over 2,048 evaluations, must exceed this (mean above 120).
pub const MIN_DISTINCT_SUM: u64 = 245_760;
/// The distinct-address bound for a program with `loads` loads per hash: the same 120 of 128 ratio, so
/// The distinct-address bound for a program with `loads` dataset loads per hash: the same 120 of 128 ratio, so
/// [`MIN_DISTINCT_SUM`] for the lottery hash and `loads x 1,920` for the read-width classes with other counts.
/// Variant 5's scratch read-modify-writes are not dataset loads: their slots repeat by design (a later
/// read-modify-write sees an earlier write), so they are neither counted nor bounded here.
pub fn min_distinct_sum(loads: usize) -> u64 {
loads as u64 * ACCEPT_HASHES as u64 * 120 / 128
}
@ -73,7 +75,7 @@ impl std::fmt::Display for Reject {
Reject::Saturated { count } => write!(f, "(c) {count} of 16384 final register values saturated (limit 163)"),
Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"),
Reject::DistinctAddresses { sum } => {
write!(f, "(c) distinct addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 loads)", *sum as f64 / 2048.0)
write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0)
}
}
}
@ -189,7 +191,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
}
let mut idx = [0u32; LANES];
let mut nload = 0usize;
let mut scratch = if p.has_scratch() { Some(ScratchModel::new()) } else { None };
let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None };
let slot_mask = p.class.scratch_slot_mask();
for it in 0..ITERATIONS {
let sel = r[0];
for (k, ins) in p.instrs.iter().enumerate() {
@ -200,7 +203,7 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
// Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word).
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
for lane in 0..LANES {
idx[lane] = r[a][lane] & SCRATCH_SLOT_MASK;
idx[lane] = r[a][lane] & slot_mask;
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
@ -332,7 +335,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
sl.sort_unstable();
let mut distinct = 0u64;
for k in 0..loads {
if k == 0 || sl[k] != sl[k - 1] {
// scratch slots carry bit 31 (variant 5) and are not dataset addresses
if sl[k] & 0x8000_0000 == 0 && (k == 0 || sl[k] != sl[k - 1]) {
distinct += 1;
}
}
@ -367,7 +371,7 @@ pub fn check_dynamic(p: &Program) -> Result<AcceptReport, Reject> {
}
bias_max = bias_max.max(d);
}
if acc.distinct_sum <= min_distinct_sum(loads) {
if acc.distinct_sum <= min_distinct_sum(loads - p.scratch_ops_per_hash()) {
return Err(Reject::DistinctAddresses { sum: acc.distinct_sum });
}
Ok(AcceptReport { distinct_sum: acc.distinct_sum, saturated: acc.saturated, bias_max })
@ -395,7 +399,7 @@ mod tests {
/// with `verify.rs` on every class (the fold is shared, the addresses are aligned the same way).
#[test]
fn classes_pass_and_match_verify() {
for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2", "scr8"] {
for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2k32", "scr8k128"] {
let c = LoadClass::parse(name).unwrap();
let p = generate_class("igneum-genesis", c);
assert!(check(&p).is_ok(), "{name}");

View file

@ -15,7 +15,6 @@ use crate::memhard::{
CACHE_TAG, CACHE_WORDS, CHACHA_ROUNDS, CHACHA_SIGMA, ITEM_ROUNDS,
};
use crate::seed::SplitMix64;
use crate::generator::{SCRATCH_BYTES_PER_WARP, SCRATCH_SLOTS, SCRATCH_SLOT_MASK, SCRATCH_WORDS_PER_LANE};
use crate::verify::{DatasetMode, DatasetSource, Epoch, FOLD_MUL, FOLD_ROT};
/// Where the words of a wide load come from (read-width experiment).
@ -127,15 +126,11 @@ fn scratch_prelude(p: &Program, dialect: CoreDialect) -> String {
CoreDialect::OpenCl => ("uint", "static inline"),
};
let mut s = String::new();
s.push_str(&format!("// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, {} slots of
", SCRATCH_SLOTS));
s.push_str("// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
");
s.push_str("// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
");
s.push_str(&format!("// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a {} KiB scratch per warp, {} slots of\n", p.class.scratch_kb, p.class.scratch_slots_per_lane()));
s.push_str("// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not\n");
s.push_str("// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.\n");
s.push_str(&format!(
"{fn_} {u} scr_fill({u} gbase, {u} lane, {u} slot, {u} j) {{ {u} sw = (j == 0u) ? {} : ((j == 1u) ? {} : {}); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }}
",
"{fn_} {u} scr_fill({u} gbase, {u} lane, {u} slot, {u} j) {{ {u} sw = (j == 0u) ? {} : ((j == 1u) ? {} : {}); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }}\n",
hex(p.seed[0]),
hex(p.seed[1]),
hex(p.seed[2])
@ -145,15 +140,14 @@ fn scratch_prelude(p: &Program, dialect: CoreDialect) -> String {
/// Variant 5: one scratch read-modify-write as a statement block. `arena`, `tag`, `gbase` and `lane` are in scope
/// (the persistent prologue). Reads 16 bytes, folds the three data words into dst, rewrites the slot behind the tag.
fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str) -> String {
fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str, slot_mask: u32) -> String {
let (u, load, store) = match dialect {
CoreDialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"),
CoreDialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"),
CoreDialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena);"),
};
format!(
"{{ {u} s_ = {a} & {}u; {load} {u} m_ = (v_.x == tag) ? 0xffffffffu : 0u; {u} w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); {u} w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); {u} w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); {u} x_ = {d} ^ w0_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w1_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w2_; {d} = x_; {store} }}",
SCRATCH_SLOT_MASK,
"{{ {u} s_ = {a} & {slot_mask}u; {load} {u} m_ = (v_.x == tag) ? 0xffffffffu : 0u; {u} w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); {u} w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); {u} w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); {u} x_ = {d} ^ w0_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w1_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w2_; {d} = x_; {store} }}",
k = hex(FOLD_MUL)
)
}
@ -162,29 +156,21 @@ fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str) -> String {
/// choice); warp `w` owns arena `w` and runs the units `w, w + N, w + 2N, ...` of the launch. Inside the loop the
/// lottery hash's text is unchanged: `gid` is the unit's first output index plus the lane. The host MUST launch
/// `groups` as a multiple of N (a uniform trip count: the OpenCL local-memory exchange carries a barrier).
fn persistent_prologue(dialect: CoreDialect) -> String {
fn persistent_prologue(dialect: CoreDialect, words_per_lane: usize) -> String {
let (u, tid, nthreads, ptr) = match dialect {
CoreDialect::Metal => ("uint", "tid", "nthreads", "device uint*"),
CoreDialect::Cuda => ("uint32_t", "(blockIdx.x * blockDim.x + threadIdx.x)", "(gridDim.x * blockDim.x)", "uint32_t*"),
CoreDialect::OpenCl => ("uint", "(uint)get_global_id(0)", "(uint)get_global_size(0)", "__global uint*"),
};
let mut s = String::new();
s.push_str(&format!(" {u} lane = {tid} & 31u;
"));
s.push_str(&format!(" {u} warp_ = {tid} >> 5;
"));
s.push_str(&format!(" {u} nwarps_ = {nthreads} >> 5;
"));
s.push_str(&format!(" {ptr} arena = scratch + ((size_t)warp_ * 32u + lane) * {}u;
", SCRATCH_WORDS_PER_LANE));
s.push_str(&format!(" for ({u} g_ = warp_; g_ < groups; g_ += nwarps_) {{
"));
s.push_str(&format!(" {u} gid = g_ * 32u + lane;
"));
s.push_str(&format!(" {u} gbase = baseNonce + g_ * 32u;
"));
s.push_str(&format!(" {u} tag = salt + g_;
"));
s.push_str(&format!(" {u} lane = {tid} & 31u;\n"));
s.push_str(&format!(" {u} warp_ = {tid} >> 5;\n"));
s.push_str(&format!(" {u} nwarps_ = {nthreads} >> 5;\n"));
s.push_str(&format!(" {ptr} arena = scratch + ((size_t)warp_ * 32u + lane) * {words_per_lane}u;\n"));
s.push_str(&format!(" for ({u} g_ = warp_; g_ < groups; g_ += nwarps_) {{\n"));
s.push_str(&format!(" {u} gid = g_ * 32u + lane;\n"));
s.push_str(&format!(" {u} gbase = baseNonce + g_ * 32u;\n"));
s.push_str(&format!(" {u} tag = salt + g_;\n"));
s
}
@ -194,20 +180,13 @@ fn scratch_header_lines(p: &Program) -> String {
return String::new();
}
let mut s = String::new();
s.push_str("// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch,
");
s.push_str("// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
");
s.push_str("#define IGNEUM_PERSISTENT_WARPS 1
");
s.push_str(&format!("#define IGNEUM_SCRATCH_OPS {} // scratch read-modify-writes per program ({} per hash)
", p.class.scratch_slots(), p.scratch_ops_per_hash()));
s.push_str(&format!("#define IGNEUM_SCRATCH_SLOTS {SCRATCH_SLOTS}u
"));
s.push_str(&format!("#define IGNEUM_SCRATCH_WORDS_PER_LANE {SCRATCH_WORDS_PER_LANE}u
"));
s.push_str(&format!("#define IGNEUM_SCRATCH_BYTES_PER_WARP {SCRATCH_BYTES_PER_WARP}u
"));
s.push_str(&format!("// Variant 5: persistent warps, a {} KiB scratch per launched warp (the host launches N warps and passes scratch,\n", p.class.scratch_kb));
s.push_str("// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).\n");
s.push_str("#define IGNEUM_PERSISTENT_WARPS 1\n");
s.push_str(&format!("#define IGNEUM_SCRATCH_OPS {} // scratch read-modify-writes per program ({} per hash)\n", p.class.scratch_slots(), p.scratch_ops_per_hash()));
s.push_str(&format!("#define IGNEUM_SCRATCH_SLOTS {}u\n", p.class.scratch_slots_per_lane()));
s.push_str(&format!("#define IGNEUM_SCRATCH_WORDS_PER_LANE {}u\n", p.class.scratch_words_per_lane()));
s.push_str(&format!("#define IGNEUM_SCRATCH_BYTES_PER_WARP {}u\n", p.class.scratch_bytes_per_warp()));
s
}
@ -472,7 +451,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound:
s.push_str(&format!(" constant uint& salt [[buffer({})]],\n", b + 2));
s.push_str(" uint tid [[thread_position_in_grid]],\n");
s.push_str(" uint nthreads [[threads_per_grid]]) {\n");
s.push_str(&persistent_prologue(CoreDialect::Metal));
s.push_str(&persistent_prologue(CoreDialect::Metal, p.class.scratch_words_per_lane()));
} else {
s.push_str(" uint gid [[thread_position_in_grid]]) {\n");
}
@ -534,7 +513,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound:
}
Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&a, false))),
Op::WLoad => format!("{d} = {d} ^ {};", fetch(word_index(&a, true))),
Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a),
Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a, p.class.scratch_slot_mask()),
};
s.push_str(&format!(" {line} // {k}\n"));
}
@ -599,7 +578,7 @@ fn cuda_instr_lines(p: &Program) -> String {
Op::Load if load_width(ins) > 1 => wide_load_stmt(CoreDialect::Cuda, &d, &a, ins.width, WideSource::Stored, None),
Op::Load => format!("{d} = {d} ^ ds[{a} & mask];"),
Op::WLoad => format!("{d} = {d} ^ ds[(__shfl_sync(0xffffffffu, {a}, 0) & wmask) + lane];"),
Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a),
Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a, p.class.scratch_slot_mask()),
};
s.push_str(&format!(" {line} // {k} {}\n", ins.op.name()));
}
@ -671,7 +650,7 @@ pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String {
let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" };
s.push_str(&format!("__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask{scratch_args}) {{\n"));
if p.has_scratch() {
s.push_str(&persistent_prologue(CoreDialect::Cuda));
s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane()));
} else {
s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n");
}
@ -792,7 +771,7 @@ pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String {
let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" };
s.push_str(&format!("__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw{scratch_args}) {{\n"));
if p.has_scratch() {
s.push_str(&persistent_prologue(CoreDialect::Cuda));
s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane()));
} else {
s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n");
}
@ -878,7 +857,7 @@ fn opencl_instr_lines(p: &Program) -> String {
Op::Load if load_width(ins) > 1 => wide_load_stmt(CoreDialect::OpenCl, &d, &a, ins.width, WideSource::Stored, None),
Op::Load => format!("{d} = {d} ^ ds[{a} & mask];"),
Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[(t_ & wmask) + lane]; }}"),
Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a),
Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a, p.class.scratch_slot_mask()),
};
s.push_str(&format!(" {line} // {k} {}\n", ins.op.name()));
}
@ -897,7 +876,7 @@ pub fn opencl_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String {
let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" };
s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw{scratch_args}) {{\n"));
if p.has_scratch() {
s.push_str(&persistent_prologue(CoreDialect::OpenCl));
s.push_str(&persistent_prologue(CoreDialect::OpenCl, p.class.scratch_words_per_lane()));
} else {
s.push_str(" uint gid = (uint)get_global_id(0);\n");
}
@ -1029,7 +1008,7 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String {
let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" };
s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask{scratch_args}) {{\n"));
if p.has_scratch() {
s.push_str(&persistent_prologue(CoreDialect::OpenCl));
s.push_str(&persistent_prologue(CoreDialect::OpenCl, p.class.scratch_words_per_lane()));
} else {
s.push_str(" uint gid = (uint)get_global_id(0);\n");
}
@ -1303,7 +1282,8 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String {
s.push_str(&format!(" \"bytes_per_hash\": {},\n", p.bytes_per_hash()));
if p.has_scratch() {
s.push_str(&format!(" \"scratch_ops_per_hash\": {},\n", p.scratch_ops_per_hash()));
s.push_str(&format!(" \"scratch\": \"variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of {SCRATCH_SLOTS} 16-byte slots per lane (lane-major); slot = src & 0x{SCRATCH_SLOT_MASK:x}; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)\",\n"));
s.push_str(&format!(" \"scratch_kib_per_warp\": {},\n", p.class.scratch_kb));
s.push_str(&format!(" \"scratch\": \"variant 5 (measurement only): persistent warps; a {kb} KiB scratch per warp of {slots} 16-byte slots per lane (lane-major); slot = src & 0x{smask:x}; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)\",\n", kb = p.class.scratch_kb, slots = p.class.scratch_slots_per_lane(), smask = p.class.scratch_slot_mask()));
}
s.push_str(&format!(" \"wide_load\": \"read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, {FOLD_ROT}) * 0x{FOLD_MUL:08x}) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots\",\n"));
}

View file

@ -170,50 +170,70 @@ pub const WIDTH_WORDS: [u8; 3] = [1, 4, 16];
pub struct LoadClass {
pub mix: [u8; 3],
pub load_slots: u8,
/// Variant 5: `Some(k)` gives the program a 1 MiB per-warp scratch (the kernels run persistent warps) and
/// turns `k` of the load slots into scratch read-modify-writes. `None` for every other class.
/// Variant 5: `Some(k)` gives the program a per-warp scratch (the kernels run persistent warps) and turns `k`
/// of the load slots into scratch read-modify-writes. `None` for every other class.
pub scratch: Option<u8>,
/// Variant 5: the scratch per warp in KiB (32 or 128; the whole working set of a card at full occupancy must
/// stay under 6 GB, coordinator's cap of 5 October 2026). 0 for every other class.
pub scratch_kb: u8,
}
/// Scratch geometry (variant 5): 2^11 slots of 16 bytes per lane (32 KiB), 32 lanes per warp (1 MiB), lane-major.
pub const SCRATCH_SLOT_BITS: u32 = 11;
pub const SCRATCH_SLOTS: usize = 1 << SCRATCH_SLOT_BITS;
pub const SCRATCH_SLOT_MASK: u32 = SCRATCH_SLOTS as u32 - 1;
pub const SCRATCH_WORDS_PER_LANE: usize = SCRATCH_SLOTS * 4;
pub const SCRATCH_BYTES_PER_WARP: usize = SCRATCH_WORDS_PER_LANE * 4 * LANES;
/// Scratch geometry (variant 5): 16-byte slots, lane-major, 32 lanes per warp; `scratch_kb` KiB per warp gives
/// `scratch_kb x 2` slots per lane (32 KiB: 64 slots, 128 KiB: 256 slots).
pub const SCRATCH_SLOT_BYTES: usize = 16;
impl LoadClass {
/// Slots per lane of the scratch (0 without one).
pub fn scratch_slots_per_lane(&self) -> usize {
self.scratch_kb as usize * 1024 / LANES / SCRATCH_SLOT_BYTES
}
pub fn scratch_slot_mask(&self) -> u32 {
self.scratch_slots_per_lane().saturating_sub(1) as u32
}
pub fn scratch_words_per_lane(&self) -> usize {
self.scratch_slots_per_lane() * 4
}
pub fn scratch_bytes_per_warp(&self) -> usize {
self.scratch_kb as usize * 1024
}
}
impl LoadClass {
/// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash.
pub const V2: LoadClass = LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None };
pub const V2: LoadClass = LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0 };
/// A fixed width (1, 4 or 16 words) with `load_slots` loads per program.
pub fn fixed(width_words: u8, load_slots: u8) -> LoadClass {
let mut mix = [0u8; 3];
let i = WIDTH_WORDS.iter().position(|&w| w == width_words).expect("width must be 1, 4 or 16 words");
mix[i] = 100;
LoadClass { mix, load_slots, scratch: None }
LoadClass { mix, load_slots, scratch: None, scratch_kb: 0 }
}
/// Per-load width drawn from `mix` (percent for 4, 16, 64 bytes), 16 loads per program.
pub fn mixed(mix: [u8; 3]) -> LoadClass {
assert_eq!(mix.iter().map(|&m| m as u32).sum::<u32>(), 100, "the mix must sum to 100");
LoadClass { mix, load_slots: LOAD_SLOTS as u8, scratch: None }
LoadClass { mix, load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0 }
}
/// Variant 5: version 2 widths, 16 memory operations of which `k` are scratch read-modify-writes.
pub fn scratch(k: u8) -> LoadClass {
/// Variant 5: version 2 widths, 16 memory operations of which `k` are scratch read-modify-writes into a
/// scratch of `kb` KiB per warp (a power of two, 1 to 128: at least one slot per lane, under the 6 GB cap).
pub fn scratch(k: u8, kb: u8) -> LoadClass {
assert!(k as usize <= LOAD_SLOTS);
LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: Some(k) }
assert!(kb.is_power_of_two() && kb <= 128, "scratch per warp must be a power of two up to 128 KiB");
LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: Some(k), scratch_kb: kb }
}
/// Parse "p4,p16,p64" or one of the names of [`LoadClass::name`].
/// Parse "p4,p16,p64" or one of the names of [`LoadClass::name`] ("scr4k32": 4 scratch ops, 32 KiB per warp).
pub fn parse(s: &str) -> Option<LoadClass> {
if let Some(k) = s.strip_prefix("scr") {
if let Some(rest) = s.strip_prefix("scr") {
let (k, kb) = rest.split_once('k')?;
let k: u8 = k.parse().ok()?;
if k as usize > LOAD_SLOTS {
let kb: u8 = kb.parse().ok()?;
if k as usize > LOAD_SLOTS || !kb.is_power_of_two() || kb > 128 {
return None;
}
return Some(LoadClass::scratch(k));
return Some(LoadClass::scratch(k, kb));
}
let (mix_s, slots) = match s.split_once("x") {
Some((m, n)) if !m.contains(',') => (m, n.parse::<u8>().ok()?),
@ -235,7 +255,7 @@ impl LoadClass {
if slots == 0 || slots as usize >= INSTR_COUNT {
return None;
}
Some(LoadClass { mix, load_slots: slots, scratch: None })
Some(LoadClass { mix, load_slots: slots, scratch: None, scratch_kb: 0 })
}
/// Scratch read-modify-writes per program (0 without a scratch).
@ -253,7 +273,7 @@ impl LoadClass {
return "v2".to_string();
}
if let Some(k) = self.scratch {
return format!("scr{k}");
return format!("scr{k}k{}", self.scratch_kb);
}
let base = match self.mix {
[100, 0, 0] => "w4".to_string(),
@ -387,6 +407,7 @@ pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &L
if let Some(k) = class.scratch {
b.extend_from_slice(b"scratch/");
b.push(k);
b.push(class.scratch_kb);
}
fnv1a64(&b)
}
@ -855,10 +876,12 @@ mod tests {
assert!(ids.insert(p.program_id()), "{name}: program id collides");
}
// variant 5: k scratch ops among the 16 memory operations, the rest one-word loads
for k in [0u8, 2, 4, 8] {
let c = LoadClass::parse(&format!("scr{k}")).unwrap();
assert_eq!(c, LoadClass::scratch(k));
assert_eq!(c.name(), format!("scr{k}"));
for (k, kb) in [(0u8, 32u8), (2, 32), (4, 128), (8, 128)] {
let c = LoadClass::parse(&format!("scr{k}k{kb}")).unwrap();
assert_eq!(c, LoadClass::scratch(k, kb));
assert_eq!(c.name(), format!("scr{k}k{kb}"));
assert_eq!(c.scratch_bytes_per_warp(), kb as usize * 1024);
assert_eq!(c.scratch_slots_per_lane(), kb as usize * 2);
let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c);
assert_eq!(p.loads_per_hash(), 128);
assert_eq!(p.scratch_ops_per_hash(), 8 * k as usize);
@ -866,7 +889,9 @@ mod tests {
assert!(p.instrs.iter().all(|i| i.width == 1));
assert!(ids.insert(p.program_id()), "scr{k}: program id collides");
}
assert!(!LoadClass::scratch(0).is_v2());
assert!(!LoadClass::scratch(0, 32).is_v2());
assert_ne!(LoadClass::scratch(4, 32).name(), LoadClass::scratch(4, 128).name());
assert_eq!(LoadClass::parse("scr4"), None);
// a class with the version 2 widths but another slot count takes the extra roll: a different stream
let w4x8 = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::fixed(1, 8));
assert_ne!(w4x8.instrs, v2.instrs);

View file

@ -1,9 +1,7 @@
//! The CPU reference interpreter for one 32-lane warp (`cpuWarpTraced` in the Swift) and the API the node
//! calls. Dataset words come from the memory-hard cache (default) or from the closed form (old packs).
use crate::generator::{
generate, generate_class, Instr, LoadClass, Op, Program, ITERATIONS, LANES, SCRATCH_SLOTS, SCRATCH_SLOT_MASK,
};
use crate::generator::{generate, generate_class, Instr, LoadClass, Op, Program, ITERATIONS, LANES};
use crate::memhard::MemhardCpu;
use crate::seed::day_key;
@ -45,9 +43,10 @@ pub fn scratch_rewrite(x: u32, w: &[u32; 3]) -> [u32; 3] {
}
/// The CPU model of one unit's scratch (variant 5): per lane, the written slots and their words. Unwritten slots
/// read as [`scratch_fill`]. A unit touches at most `scratch ops x 32` slots, so the model is small whatever the
/// nominal 1 MiB; a GPU keeps the real 1 MiB per resident warp with a per-unit tag per slot.
/// read as [`scratch_fill`]. A unit touches at most `scratch ops x 32` slots; a GPU keeps the real scratch per
/// resident warp with a per-unit tag per slot.
pub struct ScratchModel {
slots: usize,
written: Vec<bool>,
data: Vec<[u32; 3]>,
pub reads: usize,
@ -55,13 +54,19 @@ pub struct ScratchModel {
}
impl ScratchModel {
pub fn new() -> Self {
Self { written: vec![false; LANES * SCRATCH_SLOTS], data: vec![[0; 3]; LANES * SCRATCH_SLOTS], reads: 0, writes: 0 }
pub fn new(slots_per_lane: usize) -> Self {
Self {
slots: slots_per_lane,
written: vec![false; LANES * slots_per_lane],
data: vec![[0; 3]; LANES * slots_per_lane],
reads: 0,
writes: 0,
}
}
/// Read slot `slot` of `lane`, then rewrite it from the fold result `x`. Returns the three words read.
#[inline]
pub fn rmw(&mut self, seed: &[u32; 8], base: u32, lane: usize, slot: u32, dst: u32) -> u32 {
let i = lane * SCRATCH_SLOTS + slot as usize;
let i = lane * self.slots + slot as usize;
let w = if self.written[i] {
self.data[i]
} else {
@ -80,12 +85,6 @@ impl ScratchModel {
}
}
impl Default for ScratchModel {
fn default() -> Self {
Self::new()
}
}
/// Dataset element, closed form of (day words, index). The original prototype's six-operation element.
#[inline(always)]
pub fn dataset_elem(i: u32, d0: u32, d1: u32) -> u32 {
@ -258,7 +257,8 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32,
let mut items_derived = 0usize;
let mut idx = [0u32; LANES];
let mut val = [0u32; LANES];
let mut scratch = if program.has_scratch() { Some(ScratchModel::new()) } else { None };
let mut scratch = if program.has_scratch() { Some(ScratchModel::new(program.class.scratch_slots_per_lane())) } else { None };
let slot_mask = program.class.scratch_slot_mask();
for _ in 0..ITERATIONS {
let sel = r[0];
for ins in &program.instrs {
@ -267,7 +267,7 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32,
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
let (d, a) = (ins.dst as usize, ins.src as usize);
for lane in 0..LANES {
let slot = r[a][lane] & SCRATCH_SLOT_MASK;
let slot = r[a][lane] & slot_mask;
r[d][lane] = m.rmw(&program.seed, base_nonce, lane, slot, r[d][lane]);
}
}
@ -538,14 +538,14 @@ mod tests {
assert_eq!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 1));
assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2));
assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 64, 3, 100, 1));
let mut m = ScratchModel::new();
let mut m = ScratchModel::new(256);
let w = [scratch_fill(&seed, 32, 3, 100, 0), scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2)];
let x = m.rmw(&seed, 32, 3, 100, 0xabcd);
assert_eq!(x, fold_words(0xabcd, &w));
let x2 = m.rmw(&seed, 32, 3, 100, 0xabcd);
assert_eq!(x2, fold_words(0xabcd, &scratch_rewrite(x, &w)));
assert_eq!(m.reads, 2);
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::scratch(4));
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::scratch(4, 128));
assert_eq!(e.program.scratch_ops_per_hash(), 32);
assert_eq!(e.hash_warp(0), e.hash_warp(0));
}

View file

@ -25,6 +25,9 @@ typedef struct {
uint32_t datasetLog2, cacheLog2Words, cacheSegments, datasetMode, generator;
uint32_t seedw[8], keyw[8];
char seedString[600];
// read-width experiment (5 October 2026): the load class (0 when absent), bytes per hash, variant 5's scratch
uint32_t loadsPerHash, bytesPerHash, scratchOps, persistent, scratchWordsPerLane;
char loadClass[64];
// seeds.txt (or program.h): the seeds as the worker protocol carries them
char epochHex[65];
char dayHex[PF_HEX_CAP];
@ -254,6 +257,12 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) {
if (!pf_define_u32(prog, "IGNEUM_CACHE_SEGMENTS", &pk->cacheSegments)) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_CACHE_SEGMENTS"); }
if (!pf_define_u32(prog, "IGNEUM_GENERATOR", &pk->generator)) pk->generator = 1;
if (pf_define_words(prog, "IGNEUM_SEEDW_INIT", pk->seedw, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_SEEDW_INIT with 8 words"); }
pk->loadsPerHash = 128; pf_define_u32(prog, "IGNEUM_LOADS_PER_HASH", &pk->loadsPerHash);
pk->bytesPerHash = pk->loadsPerHash * 4u; pf_define_u32(prog, "IGNEUM_BYTES_PER_HASH", &pk->bytesPerHash);
pk->scratchOps = 0; pf_define_u32(prog, "IGNEUM_SCRATCH_OPS", &pk->scratchOps);
pk->persistent = 0; pf_define_u32(prog, "IGNEUM_PERSISTENT_WARPS", &pk->persistent);
pk->scratchWordsPerLane = 8192; pf_define_u32(prog, "IGNEUM_SCRATCH_WORDS_PER_LANE", &pk->scratchWordsPerLane);
strcpy(pk->loadClass, "v2"); pf_define_str(prog, "IGNEUM_LOAD_CLASS", pk->loadClass, sizeof(pk->loadClass));
if (pf_define_words(prog, "IGNEUM_KEY_INIT", pk->keyw, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_KEY_INIT with 8 words"); }
if (!pf_define_str(prog, "IGNEUM_SEED_STRING", pk->seedString, sizeof(pk->seedString))) strncpy(pk->seedString, "(no IGNEUM_SEED_STRING)", sizeof(pk->seedString) - 1);
pf_define_str(prog, "IGNEUM_SEED_BYTES_HEX", ehex, sizeof(ehex));

View file

@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;

View file

@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;

View file

@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;

View file

@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;

View file

@ -15,7 +15,7 @@
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
#define IGNEUM_GENERATOR 2
#define IGNEUM_PROGRAM_ATTEMPT 0
#define IGNEUM_PROGRAM_ID 0x2f098ae568f38029ull
#define IGNEUM_PROGRAM_ID 0xe0b70cd155c28f4bull
#define IGNEUM_DAY_STRING "2026-10-03"
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
#define IGNEUM_DAY0 0x3067619fu
@ -30,20 +30,20 @@
#define IGNEUM_OP_MIX "load=16 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 rotl=1"
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
#define IGNEUM_LOAD_CLASS "scr0"
#define IGNEUM_LOAD_CLASS "scr0k32"
#define IGNEUM_LOAD_SLOTS 16
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program
#define IGNEUM_BYTES_PER_HASH 512
#define IGNEUM_FOLD_ROT 11
#define IGNEUM_FOLD_MUL 0x9e3779b1u
// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch,
// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch,
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
#define IGNEUM_PERSISTENT_WARPS 1
#define IGNEUM_SCRATCH_OPS 0 // scratch read-modify-writes per program (0 per hash)
#define IGNEUM_SCRATCH_SLOTS 2048u
#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u
#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u
#define IGNEUM_SCRATCH_SLOTS 64u
#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u
#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
#define IGNEUM_DATASET_MODE 1

View file

@ -2,7 +2,7 @@
"format": "igneum-program-pack-3",
"generator": 2,
"attempt": 0,
"program_id": "0x2f098ae568f38029",
"program_id": "0xe0b70cd155c28f4b",
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
"dataset_mode": "memory-hard",
"seed": "igneum-genesis",
@ -15,13 +15,14 @@
"iterations": 8,
"instruction_count": 64,
"loads_per_hash": 128,
"load_class": "scr0",
"load_class": "scr0k32",
"load_slots": 16,
"load_mix_percent_4_16_64": [100, 0, 0],
"load_width_counts_4_16_64": [16, 0, 0],
"bytes_per_hash": 512,
"scratch_ops_per_hash": 0,
"scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
"scratch_kib_per_warp": 32,
"scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
"op_mix": {"load": 16, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "rotl": 1},
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",

View file

@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;

View file

@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;

View file

@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
@ -244,7 +244,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r3 = r3 ^ ds[r1 & mask]; // 31 load
r1 = r1 ^ ds[r0 & mask]; // 32 load
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load

View file

@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
@ -86,7 +86,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
@ -104,7 +104,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
r3 = r3 ^ ds[r1 & mask]; // 31 load
r1 = r1 ^ ds[r0 & mask]; // 32 load
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load

View file

@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
@ -244,7 +244,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r3 = r3 ^ ds[r1 & mask]; // 31 load
r1 = r1 ^ ds[r0 & mask]; // 32 load
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load
@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -337,7 +337,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
@ -355,7 +355,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
r3 = r3 ^ ds[r1 & mask]; // 31 load
r1 = r1 ^ ds[r0 & mask]; // 32 load
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load

View file

@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
@ -62,7 +62,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
@ -80,7 +80,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
r3 = r3 ^ ds[r1 & mask]; // 31 load
r1 = r1 ^ ds[r0 & mask]; // 32 load
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load

View file

@ -0,0 +1,67 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#ifndef IGNEUM_NO_CUDA
#include <cuda_runtime.h>
#endif
#define IGNEUM_SEED_STRING "igneum-genesis"
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
#define IGNEUM_GENERATOR 2
#define IGNEUM_PROGRAM_ATTEMPT 0
#define IGNEUM_PROGRAM_ID 0xe0afe0d155bc25d9ull
#define IGNEUM_DAY_STRING "2026-10-03"
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
#define IGNEUM_DAY0 0x3067619fu
#define IGNEUM_DAY1 0x3c269176u
#define IGNEUM_DATASET_LOG2 28
#define IGNEUM_MASK 0x0fffffffu
#define IGNEUM_LANES 32
#define IGNEUM_ITERATIONS 8
#define IGNEUM_INSTR_COUNT 64
#define IGNEUM_LOADS_PER_HASH 128
#define IGNEUM_WIDE_LOADS_PER_HASH 0
#define IGNEUM_OP_MIX "load=14 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 scratch=2 rotl=1"
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
#define IGNEUM_LOAD_CLASS "scr2k128"
#define IGNEUM_LOAD_SLOTS 16
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
#define IGNEUM_LOAD_WIDTH_COUNTS { 14, 0, 0 } // loads of 4, 16, 64 bytes per program
#define IGNEUM_BYTES_PER_HASH 448
#define IGNEUM_FOLD_ROT 11
#define IGNEUM_FOLD_MUL 0x9e3779b1u
// Variant 5: persistent warps, a 128 KiB scratch per launched warp (the host launches N warps and passes scratch,
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
#define IGNEUM_PERSISTENT_WARPS 1
#define IGNEUM_SCRATCH_OPS 2 // scratch read-modify-writes per program (16 per hash)
#define IGNEUM_SCRATCH_SLOTS 256u
#define IGNEUM_SCRATCH_WORDS_PER_LANE 1024u
#define IGNEUM_SCRATCH_BYTES_PER_WARP 131072u
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
#define IGNEUM_DATASET_MODE 1
#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }
#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u }
#define IGNEUM_CACHE_LOG2_WORDS 26
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
#define IGNEUM_CACHE_SEGMENTS 65536u
#define IGNEUM_ITEM_ROUNDS 8
#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u }
#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u }
#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u }
#ifndef IGNEUM_NO_CUDA
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt);
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
#endif

View file

@ -2,7 +2,7 @@
"format": "igneum-program-pack-3",
"generator": 2,
"attempt": 0,
"program_id": "0x2f0988e568f37cc3",
"program_id": "0xe0afe0d155bc25d9",
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
"dataset_mode": "memory-hard",
"seed": "igneum-genesis",
@ -15,13 +15,14 @@
"iterations": 8,
"instruction_count": 64,
"loads_per_hash": 128,
"load_class": "scr2",
"load_class": "scr2k128",
"load_slots": 16,
"load_mix_percent_4_16_64": [100, 0, 0],
"load_width_counts_4_16_64": [14, 0, 0],
"bytes_per_hash": 448,
"scratch_ops_per_hash": 16,
"scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
"scratch_kib_per_warp": 128,
"scratch": "variant 5 (measurement only): persistent warps; a 128 KiB scratch per warp of 256 16-byte slots per lane (lane-major); slot = src & 0xff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
"op_mix": {"load": 14, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "scratch": 2, "rotl": 1},
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",

View file

@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -70,7 +70,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
@ -88,7 +88,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
r3 = r3 ^ dataset[r1 & MASK]; // 31
r1 = r1 ^ dataset[r0 & MASK]; // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
r4 = r4 ^ dataset[r0 & MASK]; // 37

View file

@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -72,7 +72,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
@ -90,7 +90,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
r3 = r3 ^ dataset[r1 & MASK]; // 31
r1 = r1 ^ dataset[r0 & MASK]; // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
r4 = r4 ^ dataset[r0 & MASK]; // 37

View file

@ -11,22 +11,22 @@
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
{ // base nonce 0
0x66cffcc97c46e625ull, 0xbc9019f8df50fbfdull, 0x65629c90dde6016eull, 0xd647a41effa03d3bull, 0x86da3b6bbd751b99ull, 0x6ccf4240a0fb2d19ull, 0xeb39a1e06f17378cull, 0x2ea6b349b289fb10ull,
0x6067211e6c220500ull, 0x6e6095dfedd1360full, 0xbd1190d8b50e1b48ull, 0x216dc72a0c08d5b5ull, 0x5be1f8c080836b0cull, 0x2a32932a5953ed73ull, 0xcc2a3d68be83c802ull, 0xe46daca15338278full,
0xb2de43b96761e459ull, 0x9004acd06588cbeaull, 0x6a9a3543cf93004full, 0xff956d859cb6e408ull, 0x4397ec6e3c5fb045ull, 0x521dea569cd481d5ull, 0x89832b34108759f0ull, 0xf66e393836ffe4eaull,
0xb4e39af6c40ea2f4ull, 0x3adc22085dd8d648ull, 0x27efe270958bbfbbull, 0x6c80be0e8dca60d8ull, 0xa0afbc6a60260d59ull, 0x5d9a257fb9189537ull, 0xeb837aeef55dc3edull, 0xd381174dc14f8951ull
0x870ae6d97d9e85d8ull, 0x82f91989add778d3ull, 0x127e79aa060861e3ull, 0x148dec51aec6ee46ull, 0xcb5b4b144055bebaull, 0x832ea2d8305f7177ull, 0x374d405f6f35141dull, 0xe83f0b25fcc5b98cull,
0x8c6696ba39a3dfbcull, 0x5e588ceed27c0f20ull, 0x87382ec243a8208full, 0x831da1dd50dedfd7ull, 0x11af1b35d85d23faull, 0x3f203e64b5a6ae40ull, 0x5562b30447cf8941ull, 0xdbc5ddc8b06cab3cull,
0x089351d365256721ull, 0xf4e67e3e0ce8ce3dull, 0x9a6581aba1e08812ull, 0xf8c1f2017f31d0b2ull, 0x00e282fb6d67ed1bull, 0x64ceb85ec97d5368ull, 0x0db3762c566cd35full, 0x9ecf6fb65b27a141ull,
0xdfb6edb29d069ef0ull, 0xa3a2eb24fa67fb93ull, 0x2527bac1b676b544ull, 0x275704b557b5c1d5ull, 0x909fe0fbbb3d7e3eull, 0x6d3b7083f0b9dac1ull, 0xf7a58d446af0c3f0ull, 0x45a10eec5d0361ebull
},
{ // base nonce 4096
0xfb1f61aaeaeaef94ull, 0x184c2160963d8b57ull, 0x42a053c628625778ull, 0xeaad0e41c4770812ull, 0x1d6d389ceb462ce1ull, 0x4a639827672bdbd4ull, 0x857e42aa5a42dd6full, 0xb4ef399e5339979cull,
0x497a29225b099233ull, 0x71d8b42862d81954ull, 0x0af995663313bf04ull, 0xf436fd126619d7a1ull, 0x199e4e3333cff269ull, 0x64077952f3775768ull, 0x51af1d126c5e8388ull, 0xffbaf44fe6b15cfdull,
0xfc8fed86ecae34a7ull, 0x4cb548616f7a7d6bull, 0xc21d938c8b5bef35ull, 0x34789cbdd7088f71ull, 0xacb099a2c207d891ull, 0xfe1902d162374413ull, 0x26f7831c28f4020bull, 0xdf5192952b4af6b0ull,
0xecab61fe88dbaff4ull, 0x941c491f7fdb86e5ull, 0x2b1900c53f746e77ull, 0x8c40507b1caffeb2ull, 0x7532a1ec2b9169efull, 0x1cf399b0c8bfb520ull, 0xdf003d2bb8a2cc0cull, 0x4da853307fc977a9ull
0x5ba9a19be2ac506full, 0xf01aba1b9e1fbd4bull, 0x82576a1ada6a06aaull, 0x3cdfb035063961adull, 0x3b1c0146bee5cc0bull, 0xb9eb92e4388bb2edull, 0xfb3d93c5214abf98ull, 0xb624a0986997e24aull,
0x6dc096da5e72a34aull, 0x598baa91443c82dbull, 0x689d8cef7afc8df5ull, 0x41ae32225004d576ull, 0x06adbced85e30f4dull, 0x1bef955028e11da8ull, 0x8e3f1fde00391e41ull, 0xf1e29599bd9c776eull,
0xd85a42863321dfbcull, 0x2f220e8389179830ull, 0x658f1c559f2e3e28ull, 0x7ddc9adae1172cfaull, 0x493dad4ec6a7d467ull, 0x4f8cf35bbfa01901ull, 0x87971b6666cd0093ull, 0xbe71aa56ca4f56b0ull,
0xe9cee6ec93582a96ull, 0xa10e3cc76912027cull, 0x0d3b49bb042a2033ull, 0x680fa00b5d161278ull, 0xdef83e00736c287aull, 0x354ed038aae23286ull, 0xd280ce9bd9a71c97ull, 0x4f48b2b22716cd62ull
},
{ // base nonce 1000000
0x3d094bd04694b96full, 0xaecbd76cecd1a20aull, 0xbcb86febe56b17feull, 0x98082b557ba97517ull, 0xbb5f94108888564bull, 0xea3284877a30fc87ull, 0xc608fa4d5a8bb2adull, 0x946c721c511e0729ull,
0x46c6eba292083aedull, 0x936cb97231eb6795ull, 0xb1413c434c712cbbull, 0xedfd554d3948c1bdull, 0xa8a20cbef2faccd5ull, 0x5fe39d756cadbcadull, 0x208b2627380791feull, 0xf52f9374ce480218ull,
0xb9db7cd8814eb29eull, 0xf32ed2192b5a8719ull, 0x4f1b06a054940aefull, 0x406df498e4365eb5ull, 0x1982075caad345efull, 0x590f725623dbbbbdull, 0xa26d9192dedfefa5ull, 0x36219ec00da18980ull,
0x5361d0dcb0f8b1a3ull, 0x35bdefa2fbb5ffc3ull, 0xba4c2a4e473a9c80ull, 0x107d9d3030f8b9d3ull, 0xa8bb094266d6b987ull, 0x86164fdfbb1426e8ull, 0xa6e8cb895021cbbdull, 0xfe12809e9d99a243ull
0x3220aa9dc0bca592ull, 0x5409a7301dfcc3b6ull, 0x31c9b27aad8ef845ull, 0x564ceb4647002ad9ull, 0xcb7dc5fb129b07a6ull, 0xd940b224720c0393ull, 0x0e7f07a345108096ull, 0xc55a7d439b46f6a4ull,
0x0a0fab6343d5756aull, 0x397b5e178eeaa82cull, 0xe6a0adaa085838f8ull, 0x7389e3a61a07941bull, 0xeff9f511b46237bbull, 0xb35b6bf8e31ad14cull, 0x0b7d29ad2f411c1bull, 0x988e30319baf21daull,
0x362ff74de128b05bull, 0x312357fc7857b317ull, 0xb006adf1d322c446ull, 0x4dcdf54a98b429eeull, 0x18a48dc7b38023bdull, 0xb2af25e04714b6ceull, 0x4d7b97fe2f335dc2ull, 0xc161c509e57b4616ull,
0xb2b90c940eb64229ull, 0x2549b669b9e63f78ull, 0x1a1a4ff0e629079cull, 0x80ccafd83d359a54ull, 0x0232c0dfa9240d4dull, 0xc98a751590860b3full, 0x8e96a0aae858e53bull, 0x906b462113b3f109ull
}
};

View file

@ -8,22 +8,22 @@
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
"warps": [
{"base_nonce": 0, "expected": [
"0xd0f846c3cd57ae09", "0x4d5e1caf41761a5c", "0x7fc77bb7fa221a51", "0xe45b6317d4f52ae7", "0x09704a39107a3150", "0xd5166620f9dc49ab", "0xeaf1fa69e4075e55", "0x38e23d449b7616ae",
"0xf4593424c8427320", "0x9359a44a5149bd60", "0x2df93b5bc274f083", "0xfd80eb32016f8659", "0xdeb10b8a37fc3ee7", "0xc18e2ed77ab77f34", "0x7f39b618cc81c1b4", "0xdc1b7299961ad5ab",
"0x8f9c03d151013ec4", "0x668927a4a267d76f", "0x233a50c329caf635", "0x7c80406541ae3f15", "0x18fb38606c48af4a", "0x0452a913df5f112b", "0x59065cb670ba6d7d", "0xa2998e2ccd726df8",
"0xef690e221af37893", "0x4e87968e5c45903e", "0x6c6e5d4c1b35d7a9", "0x54d7e322426dcae3", "0x0d3ed6c08deda320", "0x3d6a4301fbb06a5a", "0x9eb78b04d2366566", "0x7806f8c64d2d09ff"
"0x870ae6d97d9e85d8", "0x82f91989add778d3", "0x127e79aa060861e3", "0x148dec51aec6ee46", "0xcb5b4b144055beba", "0x832ea2d8305f7177", "0x374d405f6f35141d", "0xe83f0b25fcc5b98c",
"0x8c6696ba39a3dfbc", "0x5e588ceed27c0f20", "0x87382ec243a8208f", "0x831da1dd50dedfd7", "0x11af1b35d85d23fa", "0x3f203e64b5a6ae40", "0x5562b30447cf8941", "0xdbc5ddc8b06cab3c",
"0x089351d365256721", "0xf4e67e3e0ce8ce3d", "0x9a6581aba1e08812", "0xf8c1f2017f31d0b2", "0x00e282fb6d67ed1b", "0x64ceb85ec97d5368", "0x0db3762c566cd35f", "0x9ecf6fb65b27a141",
"0xdfb6edb29d069ef0", "0xa3a2eb24fa67fb93", "0x2527bac1b676b544", "0x275704b557b5c1d5", "0x909fe0fbbb3d7e3e", "0x6d3b7083f0b9dac1", "0xf7a58d446af0c3f0", "0x45a10eec5d0361eb"
]},
{"base_nonce": 4096, "expected": [
"0xe2c97c384a85c687", "0xcf905b005ab01ebc", "0x4526799fae210c6b", "0xd4daf72ed75e8a16", "0xb3703a9e7c6820a8", "0xb6be9393fbb920bd", "0xe2fff04fb7816eb8", "0xed76ad5dbf400f91",
"0xea7d4b58ab0eb8d1", "0xf67a73b01f030532", "0x35ab823037510099", "0x5de1b1b0b3a26c69", "0xbe5dd90dbb7e632b", "0x1475918a9237e24d", "0x177fd53c45634d71", "0xa7cf00759ba28ce0",
"0xe51be0584ac3fbb4", "0x427049cc778aab35", "0x826bab125577d172", "0xd705b891b16237f5", "0xdf622fd44b180a87", "0x359398ecb79ec2de", "0x2a1e075fb078da66", "0xd480ddd8e66d26e8",
"0x07e3e86f517de466", "0xa98b2f7423557445", "0x6bab15b36bb142fe", "0xf87d147bf2cc5c0b", "0xf0294ea2b2820e03", "0xf219ac95e823d794", "0x9fa3fea85bc54264", "0xc4af3fafd2ff5201"
"0x5ba9a19be2ac506f", "0xf01aba1b9e1fbd4b", "0x82576a1ada6a06aa", "0x3cdfb035063961ad", "0x3b1c0146bee5cc0b", "0xb9eb92e4388bb2ed", "0xfb3d93c5214abf98", "0xb624a0986997e24a",
"0x6dc096da5e72a34a", "0x598baa91443c82db", "0x689d8cef7afc8df5", "0x41ae32225004d576", "0x06adbced85e30f4d", "0x1bef955028e11da8", "0x8e3f1fde00391e41", "0xf1e29599bd9c776e",
"0xd85a42863321dfbc", "0x2f220e8389179830", "0x658f1c559f2e3e28", "0x7ddc9adae1172cfa", "0x493dad4ec6a7d467", "0x4f8cf35bbfa01901", "0x87971b6666cd0093", "0xbe71aa56ca4f56b0",
"0xe9cee6ec93582a96", "0xa10e3cc76912027c", "0x0d3b49bb042a2033", "0x680fa00b5d161278", "0xdef83e00736c287a", "0x354ed038aae23286", "0xd280ce9bd9a71c97", "0x4f48b2b22716cd62"
]},
{"base_nonce": 1000000, "expected": [
"0x501f772483fac0a3", "0x461363da2c1539e0", "0x750050674de592af", "0x14ed105042cd912c", "0xc6477310878614eb", "0xe916b44e32e90e56", "0x531c92e69c2bdd73", "0x5bb0129ef7c3bd51",
"0x967ed91f7cbc7c79", "0x06fcc6a895b58b4a", "0x3bdb29fbf93cbff3", "0xbed385172d2e6abe", "0x921fc99ff4efac5e", "0x6b0090bedd9f69c7", "0x0b105c18aaa53ab0", "0xf0c4203561f3b94d",
"0x64abe33adccf7807", "0xd8f3a3b7e0242c04", "0x397da462f69fac5b", "0xe6b6b32e467b71be", "0x5318ac9e56278d04", "0xf7c5e348a4e1f5db", "0x79c994e6109646df", "0x9cdc42b0f6e82231",
"0x3f1160f02fd96ac7", "0xa69a23fc7058be21", "0xccde0a19bfdc25b6", "0xd3684a5b966f7497", "0x929db00b97a624f9", "0xfd14a40882d395a6", "0x49b0c1ecc514d6ad", "0x95d994c8349be12d"
"0x3220aa9dc0bca592", "0x5409a7301dfcc3b6", "0x31c9b27aad8ef845", "0x564ceb4647002ad9", "0xcb7dc5fb129b07a6", "0xd940b224720c0393", "0x0e7f07a345108096", "0xc55a7d439b46f6a4",
"0x0a0fab6343d5756a", "0x397b5e178eeaa82c", "0xe6a0adaa085838f8", "0x7389e3a61a07941b", "0xeff9f511b46237bb", "0xb35b6bf8e31ad14c", "0x0b7d29ad2f411c1b", "0x988e30319baf21da",
"0x362ff74de128b05b", "0x312357fc7857b317", "0xb006adf1d322c446", "0x4dcdf54a98b429ee", "0x18a48dc7b38023bd", "0xb2af25e04714b6ce", "0x4d7b97fe2f335dc2", "0xc161c509e57b4616",
"0xb2b90c940eb64229", "0x2549b669b9e63f78", "0x1a1a4ff0e629079c", "0x80ccafd83d359a54", "0x0232c0dfa9240d4d", "0xc98a751590860b3f", "0x8e96a0aae858e53b", "0x906b462113b3f109"
]}
],
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],

View file

@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
@ -242,9 +242,9 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r1 = r1 ^ ds[r0 & mask]; // 32 load
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load
@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r1 = r1 ^ ds[r4 & mask]; // 59 load
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad

View file

@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
@ -86,7 +86,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
@ -102,9 +102,9 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
r1 = r1 ^ ds[r0 & mask]; // 32 load
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load
@ -129,7 +129,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
r1 = r1 ^ ds[r4 & mask]; // 59 load
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad

View file

@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
@ -242,9 +242,9 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r1 = r1 ^ ds[r0 & mask]; // 32 load
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load
@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r1 = r1 ^ ds[r4 & mask]; // 59 load
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -337,7 +337,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
@ -353,9 +353,9 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r1 = r1 ^ ds[r0 & mask]; // 32 load
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load
@ -380,7 +380,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r1 = r1 ^ ds[r4 & mask]; // 59 load
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad

View file

@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
@ -62,7 +62,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
@ -78,9 +78,9 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
r1 = r1 ^ ds[r0 & mask]; // 32 load
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load
@ -105,7 +105,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
r1 = r1 ^ ds[r4 & mask]; // 59 load
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad

View file

@ -15,7 +15,7 @@
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
#define IGNEUM_GENERATOR 2
#define IGNEUM_PROGRAM_ATTEMPT 0
#define IGNEUM_PROGRAM_ID 0x2f0988e568f37cc3ull
#define IGNEUM_PROGRAM_ID 0xe0b080d155bd35b9ull
#define IGNEUM_DAY_STRING "2026-10-03"
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
#define IGNEUM_DAY0 0x3067619fu
@ -30,20 +30,20 @@
#define IGNEUM_OP_MIX "load=14 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 scratch=2 rotl=1"
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
#define IGNEUM_LOAD_CLASS "scr2"
#define IGNEUM_LOAD_CLASS "scr2k32"
#define IGNEUM_LOAD_SLOTS 16
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
#define IGNEUM_LOAD_WIDTH_COUNTS { 14, 0, 0 } // loads of 4, 16, 64 bytes per program
#define IGNEUM_BYTES_PER_HASH 448
#define IGNEUM_FOLD_ROT 11
#define IGNEUM_FOLD_MUL 0x9e3779b1u
// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch,
// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch,
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
#define IGNEUM_PERSISTENT_WARPS 1
#define IGNEUM_SCRATCH_OPS 2 // scratch read-modify-writes per program (16 per hash)
#define IGNEUM_SCRATCH_SLOTS 2048u
#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u
#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u
#define IGNEUM_SCRATCH_SLOTS 64u
#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u
#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
#define IGNEUM_DATASET_MODE 1

View file

@ -0,0 +1,130 @@
{
"format": "igneum-program-pack-3",
"generator": 2,
"attempt": 0,
"program_id": "0xe0b080d155bd35b9",
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
"dataset_mode": "memory-hard",
"seed": "igneum-genesis",
"seed_bytes": "69676e65756d2d67656e65736973",
"seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"],
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
"lanes": 32,
"registers": 8,
"iterations": 8,
"instruction_count": 64,
"loads_per_hash": 128,
"load_class": "scr2k32",
"load_slots": 16,
"load_mix_percent_4_16_64": [100, 0, 0],
"load_width_counts_4_16_64": [14, 0, 0],
"bytes_per_hash": 448,
"scratch_ops_per_hash": 16,
"scratch_kib_per_warp": 32,
"scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
"op_mix": {"load": 14, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "scratch": 2, "rotl": 1},
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
"op_semantics": {
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
"sub": "dst = dst - src",
"mul": "dst = dst * src (low 32)",
"mulhi": "dst = high 32 bits of dst * src",
"xor": "dst = dst ^ src",
"or": "dst = dst | src",
"rotl": "dst = rotl(dst, rot), rot in 1..31",
"rotr": "dst = rotr(dst, src & 31)",
"mad": "dst = src * src2 + dst",
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
"load": "dst = dst ^ dataset[src & dataset.mask]",
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
},
"dataset": {
"log2_words": 28,
"bytes": 1073741824,
"mask": "0x0fffffff",
"day": "2026-10-03",
"day_bytes": "6461792f323032362d31302d3033",
"day_words_from": "seed_words_from_bytes(day_bytes)",
"d0": "0x3067619f",
"d1": "0x3c269176",
"mode": "memory-hard",
"spec": "proto-metal/MEMHARD.md",
"key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"],
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s",
"word": "dataset[w] = item(w >> 4)[w & 15]"
},
"instructions": [
{"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1},
{"i": 1, "op": "add", "dst": 1, "src": 7, "src2": 2, "imm": "0x42da7657", "imm2": "0xc3bd2355", "rot": 25, "bit": 4, "mask": 16, "width": 1},
{"i": 2, "op": "add", "dst": 2, "src": 3, "src2": 0, "imm": "0x61f0b51c", "imm2": "0x2735a174", "rot": 4, "bit": 26, "mask": 2, "width": 1},
{"i": 3, "op": "mad", "dst": 4, "src": 0, "src2": 6, "imm": "0x679648a8", "imm2": "0x3044ba32", "rot": 31, "bit": 31, "mask": 4, "width": 1},
{"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1},
{"i": 5, "op": "load", "dst": 4, "src": 1, "src2": 2, "imm": "0x987c017a", "imm2": "0xf4d60559", "rot": 2, "bit": 0, "mask": 4, "width": 1},
{"i": 6, "op": "shfl", "dst": 6, "src": 3, "src2": 7, "imm": "0x6ea7b2df", "imm2": "0x9fce5071", "rot": 7, "bit": 15, "mask": 4, "width": 1},
{"i": 7, "op": "shfl", "dst": 1, "src": 5, "src2": 1, "imm": "0x26a2ecde", "imm2": "0xfec6ad22", "rot": 15, "bit": 11, "mask": 8, "width": 1},
{"i": 8, "op": "xor", "dst": 7, "src": 5, "src2": 2, "imm": "0xbe4b445c", "imm2": "0x17a5a9c7", "rot": 8, "bit": 8, "mask": 1, "width": 1},
{"i": 9, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1},
{"i": 10, "op": "or", "dst": 1, "src": 2, "src2": 3, "imm": "0x4e7dc10d", "imm2": "0x196d165c", "rot": 14, "bit": 27, "mask": 16, "width": 1},
{"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 1, "imm": "0xc5c3b55d", "imm2": "0xec061424", "rot": 26, "bit": 27, "mask": 8, "width": 1},
{"i": 12, "op": "or", "dst": 6, "src": 2, "src2": 3, "imm": "0x306542fe", "imm2": "0x1bb1b429", "rot": 31, "bit": 0, "mask": 2, "width": 1},
{"i": 13, "op": "mul", "dst": 2, "src": 5, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1},
{"i": 14, "op": "load", "dst": 1, "src": 2, "src2": 5, "imm": "0x028b4d37", "imm2": "0x7bbd78ea", "rot": 15, "bit": 2, "mask": 8, "width": 1},
{"i": 15, "op": "rotl", "dst": 7, "src": 6, "src2": 6, "imm": "0x5c88a1a7", "imm2": "0x5c628769", "rot": 1, "bit": 3, "mask": 8, "width": 1},
{"i": 16, "op": "scratch", "dst": 3, "src": 6, "src2": 7, "imm": "0xbac2ae81", "imm2": "0xcbbc7bdb", "rot": 18, "bit": 8, "mask": 8, "width": 1},
{"i": 17, "op": "load", "dst": 7, "src": 4, "src2": 2, "imm": "0xe8ab93e9", "imm2": "0xa00de107", "rot": 2, "bit": 1, "mask": 16, "width": 1},
{"i": 18, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1},
{"i": 19, "op": "mad", "dst": 4, "src": 0, "src2": 2, "imm": "0x5fba7bc2", "imm2": "0xdf099cfb", "rot": 4, "bit": 15, "mask": 16, "width": 1},
{"i": 20, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0x0a3056de", "imm2": "0x7f0c25c3", "rot": 27, "bit": 13, "mask": 8, "width": 1},
{"i": 21, "op": "xor", "dst": 5, "src": 7, "src2": 4, "imm": "0xbd066e1d", "imm2": "0x6d3ddc5a", "rot": 2, "bit": 29, "mask": 1, "width": 1},
{"i": 22, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1},
{"i": 23, "op": "load", "dst": 3, "src": 7, "src2": 2, "imm": "0xc7fcfc8f", "imm2": "0x8528b94f", "rot": 17, "bit": 13, "mask": 4, "width": 1},
{"i": 24, "op": "mulhi", "dst": 7, "src": 3, "src2": 5, "imm": "0xd91641e8", "imm2": "0xaf77faf2", "rot": 22, "bit": 21, "mask": 1, "width": 1},
{"i": 25, "op": "or", "dst": 5, "src": 4, "src2": 0, "imm": "0x84c03868", "imm2": "0xf6c691b7", "rot": 29, "bit": 14, "mask": 8, "width": 1},
{"i": 26, "op": "mad", "dst": 4, "src": 5, "src2": 2, "imm": "0x3bb2b6ba", "imm2": "0x49d95fd5", "rot": 1, "bit": 5, "mask": 8, "width": 1},
{"i": 27, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1},
{"i": 28, "op": "mulhi", "dst": 6, "src": 7, "src2": 6, "imm": "0xd69c4715", "imm2": "0xe0ebc4ce", "rot": 29, "bit": 2, "mask": 8, "width": 1},
{"i": 29, "op": "add", "dst": 6, "src": 1, "src2": 2, "imm": "0x3b2d2124", "imm2": "0x187a9128", "rot": 1, "bit": 9, "mask": 16, "width": 1},
{"i": 30, "op": "rotr", "dst": 6, "src": 7, "src2": 0, "imm": "0x5c64a589", "imm2": "0x61c9a38d", "rot": 17, "bit": 21, "mask": 16, "width": 1},
{"i": 31, "op": "load", "dst": 3, "src": 1, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1},
{"i": 32, "op": "load", "dst": 1, "src": 0, "src2": 7, "imm": "0xcc7972c4", "imm2": "0xad098d15", "rot": 30, "bit": 21, "mask": 8, "width": 1},
{"i": 33, "op": "add", "dst": 0, "src": 4, "src2": 4, "imm": "0x2c35699f", "imm2": "0x351dde38", "rot": 21, "bit": 18, "mask": 4, "width": 1},
{"i": 34, "op": "scratch", "dst": 0, "src": 2, "src2": 3, "imm": "0xfae8902b", "imm2": "0x5cd8306f", "rot": 5, "bit": 28, "mask": 16, "width": 1},
{"i": 35, "op": "mul", "dst": 0, "src": 3, "src2": 1, "imm": "0x4fa3f3db", "imm2": "0xdbf37e75", "rot": 7, "bit": 18, "mask": 4, "width": 1},
{"i": 36, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1},
{"i": 37, "op": "load", "dst": 4, "src": 0, "src2": 0, "imm": "0x04cc1d55", "imm2": "0x35c52d04", "rot": 11, "bit": 14, "mask": 2, "width": 1},
{"i": 38, "op": "mad", "dst": 1, "src": 3, "src2": 5, "imm": "0x3958f280", "imm2": "0x8713c7e1", "rot": 5, "bit": 23, "mask": 16, "width": 1},
{"i": 39, "op": "add", "dst": 0, "src": 3, "src2": 3, "imm": "0xa907b90b", "imm2": "0x1b053acf", "rot": 30, "bit": 25, "mask": 16, "width": 1},
{"i": 40, "op": "rotr", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1},
{"i": 41, "op": "mul", "dst": 3, "src": 2, "src2": 4, "imm": "0x49087d74", "imm2": "0x6348b489", "rot": 17, "bit": 9, "mask": 16, "width": 1},
{"i": 42, "op": "add", "dst": 1, "src": 5, "src2": 1, "imm": "0xa32e000c", "imm2": "0x6058c2e3", "rot": 25, "bit": 20, "mask": 8, "width": 1},
{"i": 43, "op": "xor", "dst": 3, "src": 4, "src2": 2, "imm": "0x3dad0eb6", "imm2": "0xb97578cb", "rot": 3, "bit": 27, "mask": 1, "width": 1},
{"i": 44, "op": "load", "dst": 3, "src": 5, "src2": 7, "imm": "0x374aec92", "imm2": "0x626f11df", "rot": 20, "bit": 18, "mask": 8, "width": 1},
{"i": 45, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1},
{"i": 46, "op": "xor", "dst": 7, "src": 1, "src2": 0, "imm": "0xef6ac348", "imm2": "0x963bb7e6", "rot": 26, "bit": 3, "mask": 8, "width": 1},
{"i": 47, "op": "add", "dst": 0, "src": 3, "src2": 0, "imm": "0x838b5065", "imm2": "0x36360066", "rot": 3, "bit": 31, "mask": 4, "width": 1},
{"i": 48, "op": "mulhi", "dst": 7, "src": 5, "src2": 0, "imm": "0x8458f7ac", "imm2": "0xc1c15026", "rot": 27, "bit": 15, "mask": 8, "width": 1},
{"i": 49, "op": "load", "dst": 0, "src": 2, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1},
{"i": 50, "op": "sub", "dst": 2, "src": 6, "src2": 0, "imm": "0x2baec8c9", "imm2": "0x4390f156", "rot": 3, "bit": 12, "mask": 8, "width": 1},
{"i": 51, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x19234061", "imm2": "0xe84dfade", "rot": 4, "bit": 19, "mask": 1, "width": 1},
{"i": 52, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0xdc2cd71e", "imm2": "0x1b5d334b", "rot": 9, "bit": 8, "mask": 8, "width": 1},
{"i": 53, "op": "sub", "dst": 7, "src": 0, "src2": 4, "imm": "0x605c31ec", "imm2": "0x9923ff88", "rot": 28, "bit": 25, "mask": 4, "width": 1},
{"i": 54, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1},
{"i": 55, "op": "xor", "dst": 7, "src": 5, "src2": 5, "imm": "0xad7493e7", "imm2": "0x3e400372", "rot": 13, "bit": 8, "mask": 1, "width": 1},
{"i": 56, "op": "load", "dst": 2, "src": 7, "src2": 1, "imm": "0x87e933c9", "imm2": "0x8c854c1b", "rot": 17, "bit": 3, "mask": 8, "width": 1},
{"i": 57, "op": "sub", "dst": 5, "src": 6, "src2": 5, "imm": "0x11be3bc9", "imm2": "0xbbaa8e24", "rot": 6, "bit": 5, "mask": 16, "width": 1},
{"i": 58, "op": "load", "dst": 1, "src": 3, "src2": 2, "imm": "0xa732351a", "imm2": "0xc01349cd", "rot": 14, "bit": 17, "mask": 16, "width": 1},
{"i": 59, "op": "load", "dst": 1, "src": 4, "src2": 0, "imm": "0xb20547b2", "imm2": "0xc94655de", "rot": 27, "bit": 30, "mask": 1, "width": 1},
{"i": 60, "op": "sub", "dst": 4, "src": 6, "src2": 7, "imm": "0x67cf904c", "imm2": "0x6873b216", "rot": 27, "bit": 7, "mask": 16, "width": 1},
{"i": 61, "op": "mul", "dst": 1, "src": 2, "src2": 7, "imm": "0x93ab0bf4", "imm2": "0x96158375", "rot": 14, "bit": 0, "mask": 16, "width": 1},
{"i": 62, "op": "mad", "dst": 3, "src": 6, "src2": 0, "imm": "0x41a443a3", "imm2": "0xe69d7919", "rot": 9, "bit": 0, "mask": 16, "width": 1},
{"i": 63, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x2fe0e98b", "imm2": "0xc88e2942", "rot": 5, "bit": 16, "mask": 16, "width": 1}
]
}

View file

@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -70,7 +70,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
@ -86,9 +86,9 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
r6 = rotr_var(r6, r7); // 30
r3 = r3 ^ dataset[r1 & MASK]; // 31
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
r1 = r1 ^ dataset[r0 & MASK]; // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
r4 = r4 ^ dataset[r0 & MASK]; // 37
@ -113,7 +113,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
r2 = r2 ^ dataset[r7 & MASK]; // 56
r5 = r5 - r6; // 57
r1 = r1 ^ dataset[r3 & MASK]; // 58
{ uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
r1 = r1 ^ dataset[r4 & MASK]; // 59
r4 = r4 - r6; // 60
r1 = r1 * r2; // 61
r3 = r6 * r0 + r3; // 62

View file

@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -72,7 +72,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
@ -88,9 +88,9 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
r6 = rotr_var(r6, r7); // 30
r3 = r3 ^ dataset[r1 & MASK]; // 31
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
r1 = r1 ^ dataset[r0 & MASK]; // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
r4 = r4 ^ dataset[r0 & MASK]; // 37
@ -115,7 +115,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
r2 = r2 ^ dataset[r7 & MASK]; // 56
r5 = r5 - r6; // 57
r1 = r1 ^ dataset[r3 & MASK]; // 58
{ uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
r1 = r1 ^ dataset[r4 & MASK]; // 59
r4 = r4 - r6; // 60
r1 = r1 * r2; // 61
r3 = r6 * r0 + r3; // 62

View file

@ -11,22 +11,22 @@
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
{ // base nonce 0
0xd0f846c3cd57ae09ull, 0x4d5e1caf41761a5cull, 0x7fc77bb7fa221a51ull, 0xe45b6317d4f52ae7ull, 0x09704a39107a3150ull, 0xd5166620f9dc49abull, 0xeaf1fa69e4075e55ull, 0x38e23d449b7616aeull,
0xf4593424c8427320ull, 0x9359a44a5149bd60ull, 0x2df93b5bc274f083ull, 0xfd80eb32016f8659ull, 0xdeb10b8a37fc3ee7ull, 0xc18e2ed77ab77f34ull, 0x7f39b618cc81c1b4ull, 0xdc1b7299961ad5abull,
0x8f9c03d151013ec4ull, 0x668927a4a267d76full, 0x233a50c329caf635ull, 0x7c80406541ae3f15ull, 0x18fb38606c48af4aull, 0x0452a913df5f112bull, 0x59065cb670ba6d7dull, 0xa2998e2ccd726df8ull,
0xef690e221af37893ull, 0x4e87968e5c45903eull, 0x6c6e5d4c1b35d7a9ull, 0x54d7e322426dcae3ull, 0x0d3ed6c08deda320ull, 0x3d6a4301fbb06a5aull, 0x9eb78b04d2366566ull, 0x7806f8c64d2d09ffull
0x589f62cd61c27dc1ull, 0xf6be7bab8b00a7c7ull, 0x349551c5979e330eull, 0x6d8156f5afaf0064ull, 0x81971552315d55b8ull, 0x20a3bb31ef5e202cull, 0x91818c4ec9fd5fecull, 0x35ca8cde74ac9715ull,
0xc1db0a90f80c8801ull, 0xbc52521a33053a8aull, 0x89fda590bf946dd6ull, 0xc54fe4aeaf11975full, 0xffc712960d7f4022ull, 0x4da6b3de6f9a0abdull, 0xf70ff0468e9b595aull, 0xc2d1d4434c1eb8c9ull,
0x70dc278b36920857ull, 0xfb20a2d85fa65b83ull, 0xd24316c07937542dull, 0xd505e1e3e85694bdull, 0x88c759248ae122a7ull, 0xf1c04d8ad71d5e8cull, 0x7e4d0e4e99717f12ull, 0x63d9e4b8f619ac98ull,
0xcfef2a4243ba8608ull, 0xb2908c3da3593ba9ull, 0x48bee70b2ee45b84ull, 0x8f7a09d71be2c321ull, 0x0aadd48107e05bdbull, 0x8b73ff0f34ddaaaaull, 0x9ff3a4875ecf0fe3ull, 0xaf3d701cd1cbbd4bull
},
{ // base nonce 4096
0xe2c97c384a85c687ull, 0xcf905b005ab01ebcull, 0x4526799fae210c6bull, 0xd4daf72ed75e8a16ull, 0xb3703a9e7c6820a8ull, 0xb6be9393fbb920bdull, 0xe2fff04fb7816eb8ull, 0xed76ad5dbf400f91ull,
0xea7d4b58ab0eb8d1ull, 0xf67a73b01f030532ull, 0x35ab823037510099ull, 0x5de1b1b0b3a26c69ull, 0xbe5dd90dbb7e632bull, 0x1475918a9237e24dull, 0x177fd53c45634d71ull, 0xa7cf00759ba28ce0ull,
0xe51be0584ac3fbb4ull, 0x427049cc778aab35ull, 0x826bab125577d172ull, 0xd705b891b16237f5ull, 0xdf622fd44b180a87ull, 0x359398ecb79ec2deull, 0x2a1e075fb078da66ull, 0xd480ddd8e66d26e8ull,
0x07e3e86f517de466ull, 0xa98b2f7423557445ull, 0x6bab15b36bb142feull, 0xf87d147bf2cc5c0bull, 0xf0294ea2b2820e03ull, 0xf219ac95e823d794ull, 0x9fa3fea85bc54264ull, 0xc4af3fafd2ff5201ull
0xc93652d639480287ull, 0x2bff33a7119c0ec9ull, 0x7c676a1cf9f37474ull, 0xcb1dab3b216bdd52ull, 0x236552ac169e0f4aull, 0xffb7303a10cb2833ull, 0xdf1ca9f83cf0740cull, 0xe88fcc23bd9a0a9full,
0x48118af81df77459ull, 0x49379fea0c36ec78ull, 0x931e8c0ca930cd35ull, 0xba3e0b2487710abfull, 0x56b607c6f3672398ull, 0xeb0215d7735482c3ull, 0x104b2405ae428d28ull, 0xa8c21e4eb7e2b744ull,
0xbc41f10ffa4e4621ull, 0xc51ab216e6e0a339ull, 0x285cdde4bd22e712ull, 0x4f985d36d1302ebdull, 0x82563e5cd9a31b86ull, 0x9495c481dd399662ull, 0xec6111e88d79f207ull, 0x112bf6a954166121ull,
0x1a6eda3d1a3846b6ull, 0xd25ebd17cbfaf07dull, 0xe552179181d4390full, 0xdd0129cf2d8db153ull, 0x0863f60becfbabedull, 0x1825c21e13698cecull, 0x20acd589e0408f6eull, 0x817a8413e24a68d4ull
},
{ // base nonce 1000000
0x501f772483fac0a3ull, 0x461363da2c1539e0ull, 0x750050674de592afull, 0x14ed105042cd912cull, 0xc6477310878614ebull, 0xe916b44e32e90e56ull, 0x531c92e69c2bdd73ull, 0x5bb0129ef7c3bd51ull,
0x967ed91f7cbc7c79ull, 0x06fcc6a895b58b4aull, 0x3bdb29fbf93cbff3ull, 0xbed385172d2e6abeull, 0x921fc99ff4efac5eull, 0x6b0090bedd9f69c7ull, 0x0b105c18aaa53ab0ull, 0xf0c4203561f3b94dull,
0x64abe33adccf7807ull, 0xd8f3a3b7e0242c04ull, 0x397da462f69fac5bull, 0xe6b6b32e467b71beull, 0x5318ac9e56278d04ull, 0xf7c5e348a4e1f5dbull, 0x79c994e6109646dfull, 0x9cdc42b0f6e82231ull,
0x3f1160f02fd96ac7ull, 0xa69a23fc7058be21ull, 0xccde0a19bfdc25b6ull, 0xd3684a5b966f7497ull, 0x929db00b97a624f9ull, 0xfd14a40882d395a6ull, 0x49b0c1ecc514d6adull, 0x95d994c8349be12dull
0x7535b29ea3e2823eull, 0xd84fe0e0281b6538ull, 0x14f4b14bf5185a26ull, 0x5fc4ec481d8c65bcull, 0x9ff6a4ec626c4cbcull, 0x52131ecd506117f0ull, 0x9a7db1822213f9b9ull, 0x025ac827f2f88c7cull,
0x4076da8ca02131e1ull, 0xbc9956bc70d53e0bull, 0x077cf7d297357750ull, 0xb00c8db428fbccf2ull, 0x50f16eecd1fb65c4ull, 0x4daddb3cf455583dull, 0x952b3cca95e87c92ull, 0xa7b7af6eac1a0222ull,
0xc58c0db5a99ced05ull, 0xa71a8ce697d65e94ull, 0xe29bab54459076d2ull, 0x5f613619a76c6400ull, 0xb43e9559e242a8d4ull, 0x4b5433e68aa1f302ull, 0x382b7f105840032cull, 0xbf402649fb8e9618ull,
0xf45994bb29726a41ull, 0x8a11f358c24795cbull, 0xc2c4f8902007527cull, 0xe12f65396a832dcdull, 0x307d3f495790aff0ull, 0x5fc00c5eb0c3e81eull, 0xb46600e4685191eeull, 0xce65db0e1d36f875ull
}
};

View file

@ -8,22 +8,22 @@
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
"warps": [
{"base_nonce": 0, "expected": [
"0x2273e2732203e32a", "0xa513d354bd107990", "0xe005b7515054c85f", "0x18a61b37b30cd1fb", "0xa21d5b98e8d07e9c", "0x5c24171a391d5ed0", "0x0f29e7583e1794b8", "0x7eca8a374d1a4f70",
"0x260ca011cd9ea10c", "0xce1050798fce3d43", "0x9560d939dca19041", "0x8480a8440b80ecc3", "0xfaa99aac459b739e", "0x7f083e72458e08ab", "0x78d876842f68672b", "0x3b9bcf6275d3575c",
"0x0256af61bdbf11b3", "0xefc6771cae646cbd", "0xbc44f1c9f9f54d87", "0x6caedd783487eb7d", "0x001b31fcb4fe0d4d", "0x947a7ba1057e25b6", "0xb9e5a0204d68c22a", "0x50489bed25d42661",
"0xb1019bff6d1057cd", "0xd1442990562ce940", "0xcd986a47f98801db", "0x9c8796b6df23300f", "0xbd53ad05d2c877a9", "0xc95e863774a15b0a", "0x132d8a91fb2fa67a", "0x53dd38e8eadc8a24"
"0x589f62cd61c27dc1", "0xf6be7bab8b00a7c7", "0x349551c5979e330e", "0x6d8156f5afaf0064", "0x81971552315d55b8", "0x20a3bb31ef5e202c", "0x91818c4ec9fd5fec", "0x35ca8cde74ac9715",
"0xc1db0a90f80c8801", "0xbc52521a33053a8a", "0x89fda590bf946dd6", "0xc54fe4aeaf11975f", "0xffc712960d7f4022", "0x4da6b3de6f9a0abd", "0xf70ff0468e9b595a", "0xc2d1d4434c1eb8c9",
"0x70dc278b36920857", "0xfb20a2d85fa65b83", "0xd24316c07937542d", "0xd505e1e3e85694bd", "0x88c759248ae122a7", "0xf1c04d8ad71d5e8c", "0x7e4d0e4e99717f12", "0x63d9e4b8f619ac98",
"0xcfef2a4243ba8608", "0xb2908c3da3593ba9", "0x48bee70b2ee45b84", "0x8f7a09d71be2c321", "0x0aadd48107e05bdb", "0x8b73ff0f34ddaaaa", "0x9ff3a4875ecf0fe3", "0xaf3d701cd1cbbd4b"
]},
{"base_nonce": 4096, "expected": [
"0x78c93312a03fb0ee", "0x184fea638ec9b5fb", "0x5687e8dcc4301dbf", "0xed02c94f23681dfc", "0x326d70162241ff6d", "0x452017eb4ed2dfcf", "0xc10b0e016f1e28c9", "0x691ce0cecf2a99ba",
"0x6c9506f34e0e63ce", "0x447a98c2b7fdfa40", "0x07486b0e4055b2c9", "0x41781460bd47fd5c", "0x01db316e35198291", "0xccd7e727f139a880", "0xdd7bd9efd16bf21c", "0x8285d37966656366",
"0x383deade15fe0ecb", "0x5fd64f5873c8e324", "0xad584cb6839c5e1d", "0xbb842707fb5e9460", "0x4e8bc8f87978fcbd", "0x18eb56f4a1fae881", "0x4c3b731a6b0c47a1", "0xda52cf9d69b252eb",
"0xb5ff19b2b3eeb13e", "0xe2595cea2afe42dd", "0x3ff108424c9e6e38", "0x3a8a9e1995f359ca", "0x6a6b1da662cf2126", "0x54e684c127bb181f", "0x2018caa81f1a7d50", "0x33e94d2c92d9d148"
"0xc93652d639480287", "0x2bff33a7119c0ec9", "0x7c676a1cf9f37474", "0xcb1dab3b216bdd52", "0x236552ac169e0f4a", "0xffb7303a10cb2833", "0xdf1ca9f83cf0740c", "0xe88fcc23bd9a0a9f",
"0x48118af81df77459", "0x49379fea0c36ec78", "0x931e8c0ca930cd35", "0xba3e0b2487710abf", "0x56b607c6f3672398", "0xeb0215d7735482c3", "0x104b2405ae428d28", "0xa8c21e4eb7e2b744",
"0xbc41f10ffa4e4621", "0xc51ab216e6e0a339", "0x285cdde4bd22e712", "0x4f985d36d1302ebd", "0x82563e5cd9a31b86", "0x9495c481dd399662", "0xec6111e88d79f207", "0x112bf6a954166121",
"0x1a6eda3d1a3846b6", "0xd25ebd17cbfaf07d", "0xe552179181d4390f", "0xdd0129cf2d8db153", "0x0863f60becfbabed", "0x1825c21e13698cec", "0x20acd589e0408f6e", "0x817a8413e24a68d4"
]},
{"base_nonce": 1000000, "expected": [
"0x04a41389bf3dfd3d", "0xc509164def9207df", "0x4a8ffdbdf46e429d", "0xff13bf0dc1b39aeb", "0xb852acc8e24133d7", "0x4bdd991ae56252ac", "0xa7739e74b3a054e9", "0xb4e36218d4b45fdc",
"0x8bbd323155f5edc5", "0xb7b56a90659e7fd2", "0xdff7c495b7027480", "0xffa8adb5c0302b06", "0xe97d7967d89a5672", "0x0d0c2d4e6493926e", "0xe9a5cda333cf2043", "0xdc95256d0986e5d8",
"0xd0dc211b811d6843", "0x68dfa3d0fb9a569b", "0xa9e0028dfd9178c0", "0x4a36ca1fc40b20a9", "0xe7c765c5a735294b", "0xf08954b015cb2628", "0xc69ee66ecf2740c5", "0xe3d01e899e46b089",
"0xc3558c74159c8603", "0x4c7aeb196bd01b04", "0x13c17119385f1910", "0xda7fca98e0989b8a", "0x6f95baf340817945", "0x1af52756fd3afcab", "0xb8eefc370bbe7e4b", "0x94a55055b48bd4db"
"0x7535b29ea3e2823e", "0xd84fe0e0281b6538", "0x14f4b14bf5185a26", "0x5fc4ec481d8c65bc", "0x9ff6a4ec626c4cbc", "0x52131ecd506117f0", "0x9a7db1822213f9b9", "0x025ac827f2f88c7c",
"0x4076da8ca02131e1", "0xbc9956bc70d53e0b", "0x077cf7d297357750", "0xb00c8db428fbccf2", "0x50f16eecd1fb65c4", "0x4daddb3cf455583d", "0x952b3cca95e87c92", "0xa7b7af6eac1a0222",
"0xc58c0db5a99ced05", "0xa71a8ce697d65e94", "0xe29bab54459076d2", "0x5f613619a76c6400", "0xb43e9559e242a8d4", "0x4b5433e68aa1f302", "0x382b7f105840032c", "0xbf402649fb8e9618",
"0xf45994bb29726a41", "0x8a11f358c24795cb", "0xc2c4f8902007527c", "0xe12f65396a832dcd", "0x307d3f495790aff0", "0x5fc00c5eb0c3e81e", "0xb46600e4685191ee", "0xce65db0e1d36f875"
]}
],
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],

View file

@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -214,7 +214,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
r7 = r7 ^ ds[r2 & mask]; // 4 load
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
@ -226,14 +226,14 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
{ uint s_ = r7 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
r3 = r3 ^ ds[r7 & mask]; // 23 load
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
@ -242,19 +242,19 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
r4 = r4 ^ ds[r0 & mask]; // 37 load
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint s_ = r5 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
r3 = r3 ^ ds[r5 & mask]; // 44 load
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad

View file

@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
@ -74,7 +74,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
r7 = r7 ^ ds[r2 & mask]; // 4 load
r4 = r4 ^ ds[r1 & mask]; // 5 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
@ -86,14 +86,14 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = __umulhi(r2, r5); // 22 mulhi
{ uint32_t s_ = r7 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
r3 = r3 ^ ds[r7 & mask]; // 23 load
r7 = __umulhi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
@ -102,19 +102,19 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
r4 = r4 ^ ds[r0 & mask]; // 37 load
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint32_t s_ = r5 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
r3 = r3 ^ ds[r5 & mask]; // 44 load
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
@ -129,7 +129,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
{ uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad

View file

@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -214,7 +214,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
r7 = r7 ^ ds[r2 & mask]; // 4 load
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
@ -226,14 +226,14 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
{ uint s_ = r7 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
r3 = r3 ^ ds[r7 & mask]; // 23 load
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
@ -242,19 +242,19 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
r4 = r4 ^ ds[r0 & mask]; // 37 load
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint s_ = r5 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
r3 = r3 ^ ds[r5 & mask]; // 44 load
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -325,7 +325,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
r7 = r7 ^ ds[r2 & mask]; // 4 load
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
@ -337,14 +337,14 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
{ uint s_ = r7 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
r3 = r3 ^ ds[r7 & mask]; // 23 load
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
@ -353,19 +353,19 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
r4 = r4 ^ ds[r0 & mask]; // 37 load
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint s_ = r5 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
r3 = r3 ^ ds[r5 & mask]; // 44 load
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
@ -380,7 +380,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad

View file

@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
@ -50,7 +50,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
r7 = r7 ^ ds[r2 & mask]; // 4 load
r4 = r4 ^ ds[r1 & mask]; // 5 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
@ -62,14 +62,14 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = __umulhi(r2, r5); // 22 mulhi
{ uint32_t s_ = r7 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
r3 = r3 ^ ds[r7 & mask]; // 23 load
r7 = __umulhi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
@ -78,19 +78,19 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
r4 = r4 ^ ds[r0 & mask]; // 37 load
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint32_t s_ = r5 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
r3 = r3 ^ ds[r5 & mask]; // 44 load
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
@ -105,7 +105,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
{ uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad

View file

@ -0,0 +1,67 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#ifndef IGNEUM_NO_CUDA
#include <cuda_runtime.h>
#endif
#define IGNEUM_SEED_STRING "igneum-genesis"
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
#define IGNEUM_GENERATOR 2
#define IGNEUM_PROGRAM_ATTEMPT 0
#define IGNEUM_PROGRAM_ID 0xe0c444d155cd78cfull
#define IGNEUM_DAY_STRING "2026-10-03"
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
#define IGNEUM_DAY0 0x3067619fu
#define IGNEUM_DAY1 0x3c269176u
#define IGNEUM_DATASET_LOG2 28
#define IGNEUM_MASK 0x0fffffffu
#define IGNEUM_LANES 32
#define IGNEUM_ITERATIONS 8
#define IGNEUM_INSTR_COUNT 64
#define IGNEUM_LOADS_PER_HASH 128
#define IGNEUM_WIDE_LOADS_PER_HASH 0
#define IGNEUM_OP_MIX "load=12 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 scratch=4 shfl=4 rotr=2 rotl=1"
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
#define IGNEUM_LOAD_CLASS "scr4k128"
#define IGNEUM_LOAD_SLOTS 16
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
#define IGNEUM_LOAD_WIDTH_COUNTS { 12, 0, 0 } // loads of 4, 16, 64 bytes per program
#define IGNEUM_BYTES_PER_HASH 384
#define IGNEUM_FOLD_ROT 11
#define IGNEUM_FOLD_MUL 0x9e3779b1u
// Variant 5: persistent warps, a 128 KiB scratch per launched warp (the host launches N warps and passes scratch,
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
#define IGNEUM_PERSISTENT_WARPS 1
#define IGNEUM_SCRATCH_OPS 4 // scratch read-modify-writes per program (32 per hash)
#define IGNEUM_SCRATCH_SLOTS 256u
#define IGNEUM_SCRATCH_WORDS_PER_LANE 1024u
#define IGNEUM_SCRATCH_BYTES_PER_WARP 131072u
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
#define IGNEUM_DATASET_MODE 1
#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }
#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u }
#define IGNEUM_CACHE_LOG2_WORDS 26
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
#define IGNEUM_CACHE_SEGMENTS 65536u
#define IGNEUM_ITEM_ROUNDS 8
#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u }
#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u }
#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u }
#ifndef IGNEUM_NO_CUDA
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt);
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
#endif

View file

@ -2,7 +2,7 @@
"format": "igneum-program-pack-3",
"generator": 2,
"attempt": 0,
"program_id": "0x2f098ee568f386f5",
"program_id": "0xe0c444d155cd78cf",
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
"dataset_mode": "memory-hard",
"seed": "igneum-genesis",
@ -15,13 +15,14 @@
"iterations": 8,
"instruction_count": 64,
"loads_per_hash": 128,
"load_class": "scr4",
"load_class": "scr4k128",
"load_slots": 16,
"load_mix_percent_4_16_64": [100, 0, 0],
"load_width_counts_4_16_64": [12, 0, 0],
"bytes_per_hash": 384,
"scratch_ops_per_hash": 32,
"scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
"scratch_kib_per_warp": 128,
"scratch": "variant 5 (measurement only): persistent warps; a 128 KiB scratch per warp of 256 16-byte slots per lane (lane-major); slot = src & 0xff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
"op_mix": {"load": 12, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "scratch": 4, "shfl": 4, "rotr": 2, "rotl": 1},
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",

View file

@ -0,0 +1,126 @@
#include <metal_stdlib>
using namespace metal;
#define MASK 0x0fffffffu
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
device ulong* out [[buffer(1)]],
constant uint& baseNonce [[buffer(2)]],
device uint* scratch [[buffer(3)]],
constant uint& groups [[buffer(4)]],
constant uint& salt [[buffer(5)]],
uint tid [[thread_position_in_grid]],
uint nthreads [[threads_per_grid]]) {
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
r4 = r0 * r6 + r4; // 3
r7 = r7 ^ dataset[r2 & MASK]; // 4
r4 = r4 ^ dataset[r1 & MASK]; // 5
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
r7 = r7 ^ r5; // 8
r3 = r3 | r4; // 9
r1 = r1 | r2; // 10
r4 = r4 ^ dataset[r3 & MASK]; // 11
r6 = r6 | r2; // 12
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
r5 = r5 ^ r7; // 21
r2 = mulhi(r2, r5); // 22
r3 = r3 ^ dataset[r7 & MASK]; // 23
r7 = mulhi(r7, r3); // 24
r5 = r5 | r4; // 25
r4 = r5 * r2 + r4; // 26
r5 = r5 * r1; // 27
r6 = mulhi(r6, r7); // 28
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
r6 = rotr_var(r6, r7); // 30
r3 = r3 ^ dataset[r1 & MASK]; // 31
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
r4 = r4 ^ dataset[r0 & MASK]; // 37
r1 = r3 * r5 + r1; // 38
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
r2 = rotr_var(r2, r5); // 40
r3 = r3 * r2; // 41
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
r3 = r3 ^ r4; // 43
r3 = r3 ^ dataset[r5 & MASK]; // 44
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
r7 = r7 ^ r1; // 46
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
r7 = mulhi(r7, r5); // 48
r0 = r0 ^ dataset[r2 & MASK]; // 49
r2 = r2 - r6; // 50
r7 = r7 - r5; // 51
r2 = r2 ^ r3; // 52
r7 = r7 - r0; // 53
r3 = r5 * r0 + r3; // 54
r7 = r7 ^ r5; // 55
r2 = r2 ^ dataset[r7 & MASK]; // 56
r5 = r5 - r6; // 57
r1 = r1 ^ dataset[r3 & MASK]; // 58
{ uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
r4 = r4 - r6; // 60
r1 = r1 * r2; // 61
r3 = r6 * r0 + r3; // 62
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}

View file

@ -0,0 +1,128 @@
#include <metal_stdlib>
using namespace metal;
#define MASK 0x0fffffffu
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
device ulong* out [[buffer(1)]],
constant uint& baseNonce [[buffer(2)]],
constant uint* initw [[buffer(3)]],
device uint* scratch [[buffer(4)]],
constant uint& groups [[buffer(5)]],
constant uint& salt [[buffer(6)]],
uint tid [[thread_position_in_grid]],
uint nthreads [[threads_per_grid]]) {
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
r4 = r0 * r6 + r4; // 3
r7 = r7 ^ dataset[r2 & MASK]; // 4
r4 = r4 ^ dataset[r1 & MASK]; // 5
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
r7 = r7 ^ r5; // 8
r3 = r3 | r4; // 9
r1 = r1 | r2; // 10
r4 = r4 ^ dataset[r3 & MASK]; // 11
r6 = r6 | r2; // 12
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
r5 = r5 ^ r7; // 21
r2 = mulhi(r2, r5); // 22
r3 = r3 ^ dataset[r7 & MASK]; // 23
r7 = mulhi(r7, r3); // 24
r5 = r5 | r4; // 25
r4 = r5 * r2 + r4; // 26
r5 = r5 * r1; // 27
r6 = mulhi(r6, r7); // 28
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
r6 = rotr_var(r6, r7); // 30
r3 = r3 ^ dataset[r1 & MASK]; // 31
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
r4 = r4 ^ dataset[r0 & MASK]; // 37
r1 = r3 * r5 + r1; // 38
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
r2 = rotr_var(r2, r5); // 40
r3 = r3 * r2; // 41
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
r3 = r3 ^ r4; // 43
r3 = r3 ^ dataset[r5 & MASK]; // 44
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
r7 = r7 ^ r1; // 46
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
r7 = mulhi(r7, r5); // 48
r0 = r0 ^ dataset[r2 & MASK]; // 49
r2 = r2 - r6; // 50
r7 = r7 - r5; // 51
r2 = r2 ^ r3; // 52
r7 = r7 - r0; // 53
r3 = r5 * r0 + r3; // 54
r7 = r7 ^ r5; // 55
r2 = r2 ^ dataset[r7 & MASK]; // 56
r5 = r5 - r6; // 57
r1 = r1 ^ dataset[r3 & MASK]; // 58
{ uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
r4 = r4 - r6; // 60
r1 = r1 * r2; // 61
r3 = r6 * r0 + r3; // 62
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}

View file

@ -11,22 +11,22 @@
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
{ // base nonce 0
0x2273e2732203e32aull, 0xa513d354bd107990ull, 0xe005b7515054c85full, 0x18a61b37b30cd1fbull, 0xa21d5b98e8d07e9cull, 0x5c24171a391d5ed0ull, 0x0f29e7583e1794b8ull, 0x7eca8a374d1a4f70ull,
0x260ca011cd9ea10cull, 0xce1050798fce3d43ull, 0x9560d939dca19041ull, 0x8480a8440b80ecc3ull, 0xfaa99aac459b739eull, 0x7f083e72458e08abull, 0x78d876842f68672bull, 0x3b9bcf6275d3575cull,
0x0256af61bdbf11b3ull, 0xefc6771cae646cbdull, 0xbc44f1c9f9f54d87ull, 0x6caedd783487eb7dull, 0x001b31fcb4fe0d4dull, 0x947a7ba1057e25b6ull, 0xb9e5a0204d68c22aull, 0x50489bed25d42661ull,
0xb1019bff6d1057cdull, 0xd1442990562ce940ull, 0xcd986a47f98801dbull, 0x9c8796b6df23300full, 0xbd53ad05d2c877a9ull, 0xc95e863774a15b0aull, 0x132d8a91fb2fa67aull, 0x53dd38e8eadc8a24ull
0xd48ade5043a1a440ull, 0x0236be04051c86abull, 0x4a368a9d6ce0d6eaull, 0x58518f225df7cd40ull, 0xa82cf7b3417df902ull, 0x4f2722d25e5aa401ull, 0xf94296c41a24bdb4ull, 0x4fca15537288398bull,
0x9ba276d89178f795ull, 0x7205ccbf3772e4e5ull, 0x4149e0fbf1c35bb9ull, 0x91d484090f0b09e0ull, 0xb95010c64c2d27dbull, 0x2b9fc3c52f733771ull, 0x1f2f6dd44045e10cull, 0xb78170533d4b16a6ull,
0x679056d6c824110full, 0x0f81ba5b1476c314ull, 0xeb3d83ce6ccd9cd2ull, 0x370575fe0b0e7119ull, 0x7b4ae5a0f119315full, 0x12ceac820c840cfcull, 0xd190a33169dd6d61ull, 0xf7f90f768fc7ec83ull,
0x90f11a170fce1e73ull, 0xca168b60b8a41d61ull, 0xab9da4f155dc3d4cull, 0x884d93725ea8fc2full, 0x202bf861848ba637ull, 0x7d508d345e67589eull, 0x8dfcbd9365f96ddaull, 0x5e8315d79af30be3ull
},
{ // base nonce 4096
0x78c93312a03fb0eeull, 0x184fea638ec9b5fbull, 0x5687e8dcc4301dbfull, 0xed02c94f23681dfcull, 0x326d70162241ff6dull, 0x452017eb4ed2dfcfull, 0xc10b0e016f1e28c9ull, 0x691ce0cecf2a99baull,
0x6c9506f34e0e63ceull, 0x447a98c2b7fdfa40ull, 0x07486b0e4055b2c9ull, 0x41781460bd47fd5cull, 0x01db316e35198291ull, 0xccd7e727f139a880ull, 0xdd7bd9efd16bf21cull, 0x8285d37966656366ull,
0x383deade15fe0ecbull, 0x5fd64f5873c8e324ull, 0xad584cb6839c5e1dull, 0xbb842707fb5e9460ull, 0x4e8bc8f87978fcbdull, 0x18eb56f4a1fae881ull, 0x4c3b731a6b0c47a1ull, 0xda52cf9d69b252ebull,
0xb5ff19b2b3eeb13eull, 0xe2595cea2afe42ddull, 0x3ff108424c9e6e38ull, 0x3a8a9e1995f359caull, 0x6a6b1da662cf2126ull, 0x54e684c127bb181full, 0x2018caa81f1a7d50ull, 0x33e94d2c92d9d148ull
0x0925cd0a405af0f8ull, 0x9eeae6619738a6a0ull, 0x69d83343d36fa441ull, 0xef6e0dda67f22db7ull, 0xf5a5baddc99fc6e8ull, 0x048243c3a33d6313ull, 0xa2ed984433185d72ull, 0x5ce6444f5132231eull,
0xf9c92e489ee1479bull, 0x13df97d418a1bb1dull, 0x54d8a4aa14eb5bf2ull, 0xc93ebe91c3aa1860ull, 0x12e1ca6f27af870bull, 0xa37cd8c938ec675bull, 0x0085b0d9144040d0ull, 0x389ce17c45d36ec5ull,
0xc82d6694437e2f54ull, 0xf5a7b7357bc34eb6ull, 0x5e8e9c4cdaedb41dull, 0x18bff888b957603aull, 0x661b918790cf28f7ull, 0xf1411538bea6c80full, 0x0b3f28dd15dd2a0full, 0x076cd4e3230c6857ull,
0x2cf5028d8fe8a18full, 0x270327a0333a8520ull, 0x53461f279f163486ull, 0x834a13f157378136ull, 0x5f8609aa7fbda1aeull, 0xd57be373dec55a73ull, 0xdf6b4c6134656905ull, 0x607a5b7e2a33765eull
},
{ // base nonce 1000000
0x04a41389bf3dfd3dull, 0xc509164def9207dfull, 0x4a8ffdbdf46e429dull, 0xff13bf0dc1b39aebull, 0xb852acc8e24133d7ull, 0x4bdd991ae56252acull, 0xa7739e74b3a054e9ull, 0xb4e36218d4b45fdcull,
0x8bbd323155f5edc5ull, 0xb7b56a90659e7fd2ull, 0xdff7c495b7027480ull, 0xffa8adb5c0302b06ull, 0xe97d7967d89a5672ull, 0x0d0c2d4e6493926eull, 0xe9a5cda333cf2043ull, 0xdc95256d0986e5d8ull,
0xd0dc211b811d6843ull, 0x68dfa3d0fb9a569bull, 0xa9e0028dfd9178c0ull, 0x4a36ca1fc40b20a9ull, 0xe7c765c5a735294bull, 0xf08954b015cb2628ull, 0xc69ee66ecf2740c5ull, 0xe3d01e899e46b089ull,
0xc3558c74159c8603ull, 0x4c7aeb196bd01b04ull, 0x13c17119385f1910ull, 0xda7fca98e0989b8aull, 0x6f95baf340817945ull, 0x1af52756fd3afcabull, 0xb8eefc370bbe7e4bull, 0x94a55055b48bd4dbull
0x16b8e21167f3437cull, 0xfd28ad2d0f75a03cull, 0xf8e70bdb604cfff7ull, 0xeba037043c5ece4bull, 0xa2cb7d31f4d25317ull, 0xc3e8b85a50bdab1dull, 0xd7bd65a4353eab2eull, 0x23540281fae8cce3ull,
0x37f4deac1a67cc5full, 0xd482f81bec2535a5ull, 0xc18f3f46f812b870ull, 0x582514aab0cf566dull, 0xb3b1a7424a758ac6ull, 0x83bbf70ed4151fa4ull, 0x72e2fed205f44a00ull, 0x2f81d15c1a8e17feull,
0xbe7875e7927ca851ull, 0x1a72a20d292cc17bull, 0x859dd2c75675a04bull, 0xe12711e4d81b1e04ull, 0xfafefd6afdda6b35ull, 0x30ebb4d12f5cf4e3ull, 0x6d4aae24eee724a2ull, 0x317d83e64bf5e9c6ull,
0x2beba8ecb8b0b26eull, 0x5cf2eacb7a58bd99ull, 0x2d56441aac88a037ull, 0x01708cc58adfeb95ull, 0xb3bc095b95418a2full, 0xf804c273322638e1ull, 0x9d89d48818056f24ull, 0xe07cbfd9aa54ccf1ull
}
};

View file

@ -8,22 +8,22 @@
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
"warps": [
{"base_nonce": 0, "expected": [
"0x66cffcc97c46e625", "0xbc9019f8df50fbfd", "0x65629c90dde6016e", "0xd647a41effa03d3b", "0x86da3b6bbd751b99", "0x6ccf4240a0fb2d19", "0xeb39a1e06f17378c", "0x2ea6b349b289fb10",
"0x6067211e6c220500", "0x6e6095dfedd1360f", "0xbd1190d8b50e1b48", "0x216dc72a0c08d5b5", "0x5be1f8c080836b0c", "0x2a32932a5953ed73", "0xcc2a3d68be83c802", "0xe46daca15338278f",
"0xb2de43b96761e459", "0x9004acd06588cbea", "0x6a9a3543cf93004f", "0xff956d859cb6e408", "0x4397ec6e3c5fb045", "0x521dea569cd481d5", "0x89832b34108759f0", "0xf66e393836ffe4ea",
"0xb4e39af6c40ea2f4", "0x3adc22085dd8d648", "0x27efe270958bbfbb", "0x6c80be0e8dca60d8", "0xa0afbc6a60260d59", "0x5d9a257fb9189537", "0xeb837aeef55dc3ed", "0xd381174dc14f8951"
"0xd48ade5043a1a440", "0x0236be04051c86ab", "0x4a368a9d6ce0d6ea", "0x58518f225df7cd40", "0xa82cf7b3417df902", "0x4f2722d25e5aa401", "0xf94296c41a24bdb4", "0x4fca15537288398b",
"0x9ba276d89178f795", "0x7205ccbf3772e4e5", "0x4149e0fbf1c35bb9", "0x91d484090f0b09e0", "0xb95010c64c2d27db", "0x2b9fc3c52f733771", "0x1f2f6dd44045e10c", "0xb78170533d4b16a6",
"0x679056d6c824110f", "0x0f81ba5b1476c314", "0xeb3d83ce6ccd9cd2", "0x370575fe0b0e7119", "0x7b4ae5a0f119315f", "0x12ceac820c840cfc", "0xd190a33169dd6d61", "0xf7f90f768fc7ec83",
"0x90f11a170fce1e73", "0xca168b60b8a41d61", "0xab9da4f155dc3d4c", "0x884d93725ea8fc2f", "0x202bf861848ba637", "0x7d508d345e67589e", "0x8dfcbd9365f96dda", "0x5e8315d79af30be3"
]},
{"base_nonce": 4096, "expected": [
"0xfb1f61aaeaeaef94", "0x184c2160963d8b57", "0x42a053c628625778", "0xeaad0e41c4770812", "0x1d6d389ceb462ce1", "0x4a639827672bdbd4", "0x857e42aa5a42dd6f", "0xb4ef399e5339979c",
"0x497a29225b099233", "0x71d8b42862d81954", "0x0af995663313bf04", "0xf436fd126619d7a1", "0x199e4e3333cff269", "0x64077952f3775768", "0x51af1d126c5e8388", "0xffbaf44fe6b15cfd",
"0xfc8fed86ecae34a7", "0x4cb548616f7a7d6b", "0xc21d938c8b5bef35", "0x34789cbdd7088f71", "0xacb099a2c207d891", "0xfe1902d162374413", "0x26f7831c28f4020b", "0xdf5192952b4af6b0",
"0xecab61fe88dbaff4", "0x941c491f7fdb86e5", "0x2b1900c53f746e77", "0x8c40507b1caffeb2", "0x7532a1ec2b9169ef", "0x1cf399b0c8bfb520", "0xdf003d2bb8a2cc0c", "0x4da853307fc977a9"
"0x0925cd0a405af0f8", "0x9eeae6619738a6a0", "0x69d83343d36fa441", "0xef6e0dda67f22db7", "0xf5a5baddc99fc6e8", "0x048243c3a33d6313", "0xa2ed984433185d72", "0x5ce6444f5132231e",
"0xf9c92e489ee1479b", "0x13df97d418a1bb1d", "0x54d8a4aa14eb5bf2", "0xc93ebe91c3aa1860", "0x12e1ca6f27af870b", "0xa37cd8c938ec675b", "0x0085b0d9144040d0", "0x389ce17c45d36ec5",
"0xc82d6694437e2f54", "0xf5a7b7357bc34eb6", "0x5e8e9c4cdaedb41d", "0x18bff888b957603a", "0x661b918790cf28f7", "0xf1411538bea6c80f", "0x0b3f28dd15dd2a0f", "0x076cd4e3230c6857",
"0x2cf5028d8fe8a18f", "0x270327a0333a8520", "0x53461f279f163486", "0x834a13f157378136", "0x5f8609aa7fbda1ae", "0xd57be373dec55a73", "0xdf6b4c6134656905", "0x607a5b7e2a33765e"
]},
{"base_nonce": 1000000, "expected": [
"0x3d094bd04694b96f", "0xaecbd76cecd1a20a", "0xbcb86febe56b17fe", "0x98082b557ba97517", "0xbb5f94108888564b", "0xea3284877a30fc87", "0xc608fa4d5a8bb2ad", "0x946c721c511e0729",
"0x46c6eba292083aed", "0x936cb97231eb6795", "0xb1413c434c712cbb", "0xedfd554d3948c1bd", "0xa8a20cbef2faccd5", "0x5fe39d756cadbcad", "0x208b2627380791fe", "0xf52f9374ce480218",
"0xb9db7cd8814eb29e", "0xf32ed2192b5a8719", "0x4f1b06a054940aef", "0x406df498e4365eb5", "0x1982075caad345ef", "0x590f725623dbbbbd", "0xa26d9192dedfefa5", "0x36219ec00da18980",
"0x5361d0dcb0f8b1a3", "0x35bdefa2fbb5ffc3", "0xba4c2a4e473a9c80", "0x107d9d3030f8b9d3", "0xa8bb094266d6b987", "0x86164fdfbb1426e8", "0xa6e8cb895021cbbd", "0xfe12809e9d99a243"
"0x16b8e21167f3437c", "0xfd28ad2d0f75a03c", "0xf8e70bdb604cfff7", "0xeba037043c5ece4b", "0xa2cb7d31f4d25317", "0xc3e8b85a50bdab1d", "0xd7bd65a4353eab2e", "0x23540281fae8cce3",
"0x37f4deac1a67cc5f", "0xd482f81bec2535a5", "0xc18f3f46f812b870", "0x582514aab0cf566d", "0xb3b1a7424a758ac6", "0x83bbf70ed4151fa4", "0x72e2fed205f44a00", "0x2f81d15c1a8e17fe",
"0xbe7875e7927ca851", "0x1a72a20d292cc17b", "0x859dd2c75675a04b", "0xe12711e4d81b1e04", "0xfafefd6afdda6b35", "0x30ebb4d12f5cf4e3", "0x6d4aae24eee724a2", "0x317d83e64bf5e9c6",
"0x2beba8ecb8b0b26e", "0x5cf2eacb7a58bd99", "0x2d56441aac88a037", "0x01708cc58adfeb95", "0xb3bc095b95418a2f", "0xf804c273322638e1", "0x9d89d48818056f24", "0xe07cbfd9aa54ccf1"
]}
],
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],

View file

@ -0,0 +1,291 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
#ifndef IGNEUM_GROUP
#define IGNEUM_GROUP 32
#endif
#ifndef IGNEUM_EXCHANGE
#define IGNEUM_EXCHANGE 0
#endif
#ifdef __OPENCL_VERSION__
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
#if IGNEUM_EXCHANGE == 1
#ifdef cl_khr_subgroups
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
#endif
#ifdef cl_khr_subgroup_shuffle
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
#endif
#elif IGNEUM_EXCHANGE == 2
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
#endif
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
#else
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
#include "emu_opencl.h"
#endif
#if IGNEUM_EXCHANGE == 1
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#elif IGNEUM_EXCHANGE == 2
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#else
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
#endif
static inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
static inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
static inline void mh_chacha_block(const uint* x, uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
static inline void mh_cache_segment(__global uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
static inline void mh_mixer(uint* s, uint rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
// The same constants as memhard.h in this pack (one emitter, three dialects).
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
uint seg = (uint)get_global_id(0);
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
uint t = (uint)get_global_id(0);
if (t < nItems) {
uint s[16];
mh_item(cache, t, s);
__global uint* d = ds + ((ulong)t * 16u);
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
}
}
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
r7 = r7 ^ ds[r2 & mask]; // 4 load
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
r3 = r3 ^ ds[r7 & mask]; // 23 load
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = mul_hi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
r3 = r3 ^ ds[r5 & mask]; // 44 load
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = mul_hi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}
#if IGNEUM_EXCHANGE != 0
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
}
#endif

View file

@ -0,0 +1,177 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
#include "memhard.h"
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
uint32_t x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
if (t < nItems) {
uint32_t s[16];
mh_item(cache, t, s);
uint32_t* d = ds + (size_t)t * 16u;
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
}
}
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, uint32_t* scratch, uint32_t groups, uint32_t salt) {
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
uint32_t tag = salt + g_;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
{ uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint32_t it = 0u; it < 8u; ++it) {
uint32_t sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
r7 = r7 ^ ds[r2 & mask]; // 4 load
r4 = r4 ^ ds[r1 & mask]; // 5 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = __umulhi(r2, r5); // 22 mulhi
r3 = r3 ^ ds[r7 & mask]; // 23 load
r7 = __umulhi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = __umulhi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
r3 = r3 ^ ds[r5 & mask]; // 44 load
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = __umulhi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
}
// Host-side launch wrappers. Declared in program.h, called from host.cu.
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
if (nSegments == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nSegments + block - 1u) / block;
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
return cudaGetLastError();
}
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
if (nItems == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nItems + block - 1u) / block;
igneum_build<<<grid, block>>>(ds, cache, nItems);
return cudaGetLastError();
}
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
igneum_hash<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, scratch, nonces / 32u, salt);
return cudaGetLastError();
}
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
}

View file

@ -0,0 +1,393 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
#ifndef IGNEUM_GROUP
#define IGNEUM_GROUP 32
#endif
#ifndef IGNEUM_EXCHANGE
#define IGNEUM_EXCHANGE 0
#endif
#ifdef __OPENCL_VERSION__
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
#if IGNEUM_EXCHANGE == 1
#ifdef cl_khr_subgroups
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
#endif
#ifdef cl_khr_subgroup_shuffle
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
#endif
#elif IGNEUM_EXCHANGE == 2
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
#endif
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
#else
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
#include "emu_opencl.h"
#endif
#if IGNEUM_EXCHANGE == 1
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#elif IGNEUM_EXCHANGE == 2
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#else
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
#endif
static inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
static inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
static inline void mh_chacha_block(const uint* x, uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
static inline void mh_cache_segment(__global uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
static inline void mh_mixer(uint* s, uint rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
// The same constants as memhard.h in this pack (one emitter, three dialects).
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
uint seg = (uint)get_global_id(0);
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
uint t = (uint)get_global_id(0);
if (t < nItems) {
uint s[16];
mh_item(cache, t, s);
__global uint* d = ds + ((ulong)t * 16u);
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
}
}
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
r7 = r7 ^ ds[r2 & mask]; // 4 load
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
r3 = r3 ^ ds[r7 & mask]; // 23 load
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = mul_hi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
r3 = r3 ^ ds[r5 & mask]; // 44 load
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = mul_hi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}
#if IGNEUM_EXCHANGE != 0
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
}
#endif
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global uint* scratch, uint groups, uint salt) {
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
r7 = r7 ^ ds[r2 & mask]; // 4 load
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
r3 = r3 ^ ds[r7 & mask]; // 23 load
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = mul_hi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
r3 = r3 ^ ds[r5 & mask]; // 44 load
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = mul_hi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}

View file

@ -0,0 +1,136 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
// Host declarations (also in program_bound.h if present):
// struct IgneumInitWords { uint32_t w[8]; };
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
struct IgneumInitWords { uint32_t w[8]; };
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, uint32_t* scratch, uint32_t groups, uint32_t salt) {
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
uint32_t tag = salt + g_;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
for (uint32_t it = 0u; it < 8u; ++it) {
uint32_t sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
r7 = r7 ^ ds[r2 & mask]; // 4 load
r4 = r4 ^ ds[r1 & mask]; // 5 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = __umulhi(r2, r5); // 22 mulhi
r3 = r3 ^ ds[r7 & mask]; // 23 load
r7 = __umulhi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = __umulhi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
r4 = r4 ^ ds[r0 & mask]; // 37 load
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
r3 = r3 ^ ds[r5 & mask]; // 44 load
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = __umulhi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
}
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
igneum_hash_bound<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, iw, scratch, nonces / 32u, salt);
return cudaGetLastError();
}
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
}

View file

@ -0,0 +1,108 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#if defined(__CUDACC__)
#define IGNEUM_HD __host__ __device__ __forceinline__
#elif defined(_MSC_VER) && !defined(__cplusplus)
#define IGNEUM_HD static __inline
#else
#define IGNEUM_HD static inline
#endif
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint32_t r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint32_t r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }

View file

@ -0,0 +1,106 @@
#include <metal_stdlib>
using namespace metal;
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
inline void mh_cache_segment(device uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
inline void mh_mixer(thread uint* s, uint rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
// One thread per segment (2^16 threads).
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
mh_cache_segment(cache, gid);
}
// One thread per 64-byte item (dataset words / 16 threads).
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
uint gid [[thread_position_in_grid]]) {
uint s[16];
mh_item(cache, gid, s);
device uint* d = dataset + gid * 16u;
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
}

View file

@ -15,7 +15,7 @@
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
#define IGNEUM_GENERATOR 2
#define IGNEUM_PROGRAM_ATTEMPT 0
#define IGNEUM_PROGRAM_ID 0x2f098ee568f386f5ull
#define IGNEUM_PROGRAM_ID 0xe0c4a4d155ce1befull
#define IGNEUM_DAY_STRING "2026-10-03"
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
#define IGNEUM_DAY0 0x3067619fu
@ -30,20 +30,20 @@
#define IGNEUM_OP_MIX "load=12 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 scratch=4 shfl=4 rotr=2 rotl=1"
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
#define IGNEUM_LOAD_CLASS "scr4"
#define IGNEUM_LOAD_CLASS "scr4k32"
#define IGNEUM_LOAD_SLOTS 16
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
#define IGNEUM_LOAD_WIDTH_COUNTS { 12, 0, 0 } // loads of 4, 16, 64 bytes per program
#define IGNEUM_BYTES_PER_HASH 384
#define IGNEUM_FOLD_ROT 11
#define IGNEUM_FOLD_MUL 0x9e3779b1u
// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch,
// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch,
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
#define IGNEUM_PERSISTENT_WARPS 1
#define IGNEUM_SCRATCH_OPS 4 // scratch read-modify-writes per program (32 per hash)
#define IGNEUM_SCRATCH_SLOTS 2048u
#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u
#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u
#define IGNEUM_SCRATCH_SLOTS 64u
#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u
#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
#define IGNEUM_DATASET_MODE 1

View file

@ -0,0 +1,130 @@
{
"format": "igneum-program-pack-3",
"generator": 2,
"attempt": 0,
"program_id": "0xe0c4a4d155ce1bef",
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
"dataset_mode": "memory-hard",
"seed": "igneum-genesis",
"seed_bytes": "69676e65756d2d67656e65736973",
"seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"],
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
"lanes": 32,
"registers": 8,
"iterations": 8,
"instruction_count": 64,
"loads_per_hash": 128,
"load_class": "scr4k32",
"load_slots": 16,
"load_mix_percent_4_16_64": [100, 0, 0],
"load_width_counts_4_16_64": [12, 0, 0],
"bytes_per_hash": 384,
"scratch_ops_per_hash": 32,
"scratch_kib_per_warp": 32,
"scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
"op_mix": {"load": 12, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "scratch": 4, "shfl": 4, "rotr": 2, "rotl": 1},
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
"op_semantics": {
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
"sub": "dst = dst - src",
"mul": "dst = dst * src (low 32)",
"mulhi": "dst = high 32 bits of dst * src",
"xor": "dst = dst ^ src",
"or": "dst = dst | src",
"rotl": "dst = rotl(dst, rot), rot in 1..31",
"rotr": "dst = rotr(dst, src & 31)",
"mad": "dst = src * src2 + dst",
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
"load": "dst = dst ^ dataset[src & dataset.mask]",
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
},
"dataset": {
"log2_words": 28,
"bytes": 1073741824,
"mask": "0x0fffffff",
"day": "2026-10-03",
"day_bytes": "6461792f323032362d31302d3033",
"day_words_from": "seed_words_from_bytes(day_bytes)",
"d0": "0x3067619f",
"d1": "0x3c269176",
"mode": "memory-hard",
"spec": "proto-metal/MEMHARD.md",
"key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"],
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s",
"word": "dataset[w] = item(w >> 4)[w & 15]"
},
"instructions": [
{"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1},
{"i": 1, "op": "add", "dst": 1, "src": 7, "src2": 2, "imm": "0x42da7657", "imm2": "0xc3bd2355", "rot": 25, "bit": 4, "mask": 16, "width": 1},
{"i": 2, "op": "add", "dst": 2, "src": 3, "src2": 0, "imm": "0x61f0b51c", "imm2": "0x2735a174", "rot": 4, "bit": 26, "mask": 2, "width": 1},
{"i": 3, "op": "mad", "dst": 4, "src": 0, "src2": 6, "imm": "0x679648a8", "imm2": "0x3044ba32", "rot": 31, "bit": 31, "mask": 4, "width": 1},
{"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1},
{"i": 5, "op": "load", "dst": 4, "src": 1, "src2": 2, "imm": "0x987c017a", "imm2": "0xf4d60559", "rot": 2, "bit": 0, "mask": 4, "width": 1},
{"i": 6, "op": "shfl", "dst": 6, "src": 3, "src2": 7, "imm": "0x6ea7b2df", "imm2": "0x9fce5071", "rot": 7, "bit": 15, "mask": 4, "width": 1},
{"i": 7, "op": "shfl", "dst": 1, "src": 5, "src2": 1, "imm": "0x26a2ecde", "imm2": "0xfec6ad22", "rot": 15, "bit": 11, "mask": 8, "width": 1},
{"i": 8, "op": "xor", "dst": 7, "src": 5, "src2": 2, "imm": "0xbe4b445c", "imm2": "0x17a5a9c7", "rot": 8, "bit": 8, "mask": 1, "width": 1},
{"i": 9, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1},
{"i": 10, "op": "or", "dst": 1, "src": 2, "src2": 3, "imm": "0x4e7dc10d", "imm2": "0x196d165c", "rot": 14, "bit": 27, "mask": 16, "width": 1},
{"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 1, "imm": "0xc5c3b55d", "imm2": "0xec061424", "rot": 26, "bit": 27, "mask": 8, "width": 1},
{"i": 12, "op": "or", "dst": 6, "src": 2, "src2": 3, "imm": "0x306542fe", "imm2": "0x1bb1b429", "rot": 31, "bit": 0, "mask": 2, "width": 1},
{"i": 13, "op": "mul", "dst": 2, "src": 5, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1},
{"i": 14, "op": "load", "dst": 1, "src": 2, "src2": 5, "imm": "0x028b4d37", "imm2": "0x7bbd78ea", "rot": 15, "bit": 2, "mask": 8, "width": 1},
{"i": 15, "op": "rotl", "dst": 7, "src": 6, "src2": 6, "imm": "0x5c88a1a7", "imm2": "0x5c628769", "rot": 1, "bit": 3, "mask": 8, "width": 1},
{"i": 16, "op": "scratch", "dst": 3, "src": 6, "src2": 7, "imm": "0xbac2ae81", "imm2": "0xcbbc7bdb", "rot": 18, "bit": 8, "mask": 8, "width": 1},
{"i": 17, "op": "load", "dst": 7, "src": 4, "src2": 2, "imm": "0xe8ab93e9", "imm2": "0xa00de107", "rot": 2, "bit": 1, "mask": 16, "width": 1},
{"i": 18, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1},
{"i": 19, "op": "mad", "dst": 4, "src": 0, "src2": 2, "imm": "0x5fba7bc2", "imm2": "0xdf099cfb", "rot": 4, "bit": 15, "mask": 16, "width": 1},
{"i": 20, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0x0a3056de", "imm2": "0x7f0c25c3", "rot": 27, "bit": 13, "mask": 8, "width": 1},
{"i": 21, "op": "xor", "dst": 5, "src": 7, "src2": 4, "imm": "0xbd066e1d", "imm2": "0x6d3ddc5a", "rot": 2, "bit": 29, "mask": 1, "width": 1},
{"i": 22, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1},
{"i": 23, "op": "load", "dst": 3, "src": 7, "src2": 2, "imm": "0xc7fcfc8f", "imm2": "0x8528b94f", "rot": 17, "bit": 13, "mask": 4, "width": 1},
{"i": 24, "op": "mulhi", "dst": 7, "src": 3, "src2": 5, "imm": "0xd91641e8", "imm2": "0xaf77faf2", "rot": 22, "bit": 21, "mask": 1, "width": 1},
{"i": 25, "op": "or", "dst": 5, "src": 4, "src2": 0, "imm": "0x84c03868", "imm2": "0xf6c691b7", "rot": 29, "bit": 14, "mask": 8, "width": 1},
{"i": 26, "op": "mad", "dst": 4, "src": 5, "src2": 2, "imm": "0x3bb2b6ba", "imm2": "0x49d95fd5", "rot": 1, "bit": 5, "mask": 8, "width": 1},
{"i": 27, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1},
{"i": 28, "op": "mulhi", "dst": 6, "src": 7, "src2": 6, "imm": "0xd69c4715", "imm2": "0xe0ebc4ce", "rot": 29, "bit": 2, "mask": 8, "width": 1},
{"i": 29, "op": "add", "dst": 6, "src": 1, "src2": 2, "imm": "0x3b2d2124", "imm2": "0x187a9128", "rot": 1, "bit": 9, "mask": 16, "width": 1},
{"i": 30, "op": "rotr", "dst": 6, "src": 7, "src2": 0, "imm": "0x5c64a589", "imm2": "0x61c9a38d", "rot": 17, "bit": 21, "mask": 16, "width": 1},
{"i": 31, "op": "load", "dst": 3, "src": 1, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1},
{"i": 32, "op": "scratch", "dst": 1, "src": 0, "src2": 7, "imm": "0xcc7972c4", "imm2": "0xad098d15", "rot": 30, "bit": 21, "mask": 8, "width": 1},
{"i": 33, "op": "add", "dst": 0, "src": 4, "src2": 4, "imm": "0x2c35699f", "imm2": "0x351dde38", "rot": 21, "bit": 18, "mask": 4, "width": 1},
{"i": 34, "op": "scratch", "dst": 0, "src": 2, "src2": 3, "imm": "0xfae8902b", "imm2": "0x5cd8306f", "rot": 5, "bit": 28, "mask": 16, "width": 1},
{"i": 35, "op": "mul", "dst": 0, "src": 3, "src2": 1, "imm": "0x4fa3f3db", "imm2": "0xdbf37e75", "rot": 7, "bit": 18, "mask": 4, "width": 1},
{"i": 36, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1},
{"i": 37, "op": "load", "dst": 4, "src": 0, "src2": 0, "imm": "0x04cc1d55", "imm2": "0x35c52d04", "rot": 11, "bit": 14, "mask": 2, "width": 1},
{"i": 38, "op": "mad", "dst": 1, "src": 3, "src2": 5, "imm": "0x3958f280", "imm2": "0x8713c7e1", "rot": 5, "bit": 23, "mask": 16, "width": 1},
{"i": 39, "op": "add", "dst": 0, "src": 3, "src2": 3, "imm": "0xa907b90b", "imm2": "0x1b053acf", "rot": 30, "bit": 25, "mask": 16, "width": 1},
{"i": 40, "op": "rotr", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1},
{"i": 41, "op": "mul", "dst": 3, "src": 2, "src2": 4, "imm": "0x49087d74", "imm2": "0x6348b489", "rot": 17, "bit": 9, "mask": 16, "width": 1},
{"i": 42, "op": "add", "dst": 1, "src": 5, "src2": 1, "imm": "0xa32e000c", "imm2": "0x6058c2e3", "rot": 25, "bit": 20, "mask": 8, "width": 1},
{"i": 43, "op": "xor", "dst": 3, "src": 4, "src2": 2, "imm": "0x3dad0eb6", "imm2": "0xb97578cb", "rot": 3, "bit": 27, "mask": 1, "width": 1},
{"i": 44, "op": "load", "dst": 3, "src": 5, "src2": 7, "imm": "0x374aec92", "imm2": "0x626f11df", "rot": 20, "bit": 18, "mask": 8, "width": 1},
{"i": 45, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1},
{"i": 46, "op": "xor", "dst": 7, "src": 1, "src2": 0, "imm": "0xef6ac348", "imm2": "0x963bb7e6", "rot": 26, "bit": 3, "mask": 8, "width": 1},
{"i": 47, "op": "add", "dst": 0, "src": 3, "src2": 0, "imm": "0x838b5065", "imm2": "0x36360066", "rot": 3, "bit": 31, "mask": 4, "width": 1},
{"i": 48, "op": "mulhi", "dst": 7, "src": 5, "src2": 0, "imm": "0x8458f7ac", "imm2": "0xc1c15026", "rot": 27, "bit": 15, "mask": 8, "width": 1},
{"i": 49, "op": "load", "dst": 0, "src": 2, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1},
{"i": 50, "op": "sub", "dst": 2, "src": 6, "src2": 0, "imm": "0x2baec8c9", "imm2": "0x4390f156", "rot": 3, "bit": 12, "mask": 8, "width": 1},
{"i": 51, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x19234061", "imm2": "0xe84dfade", "rot": 4, "bit": 19, "mask": 1, "width": 1},
{"i": 52, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0xdc2cd71e", "imm2": "0x1b5d334b", "rot": 9, "bit": 8, "mask": 8, "width": 1},
{"i": 53, "op": "sub", "dst": 7, "src": 0, "src2": 4, "imm": "0x605c31ec", "imm2": "0x9923ff88", "rot": 28, "bit": 25, "mask": 4, "width": 1},
{"i": 54, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1},
{"i": 55, "op": "xor", "dst": 7, "src": 5, "src2": 5, "imm": "0xad7493e7", "imm2": "0x3e400372", "rot": 13, "bit": 8, "mask": 1, "width": 1},
{"i": 56, "op": "load", "dst": 2, "src": 7, "src2": 1, "imm": "0x87e933c9", "imm2": "0x8c854c1b", "rot": 17, "bit": 3, "mask": 8, "width": 1},
{"i": 57, "op": "sub", "dst": 5, "src": 6, "src2": 5, "imm": "0x11be3bc9", "imm2": "0xbbaa8e24", "rot": 6, "bit": 5, "mask": 16, "width": 1},
{"i": 58, "op": "load", "dst": 1, "src": 3, "src2": 2, "imm": "0xa732351a", "imm2": "0xc01349cd", "rot": 14, "bit": 17, "mask": 16, "width": 1},
{"i": 59, "op": "scratch", "dst": 1, "src": 4, "src2": 0, "imm": "0xb20547b2", "imm2": "0xc94655de", "rot": 27, "bit": 30, "mask": 1, "width": 1},
{"i": 60, "op": "sub", "dst": 4, "src": 6, "src2": 7, "imm": "0x67cf904c", "imm2": "0x6873b216", "rot": 27, "bit": 7, "mask": 16, "width": 1},
{"i": 61, "op": "mul", "dst": 1, "src": 2, "src2": 7, "imm": "0x93ab0bf4", "imm2": "0x96158375", "rot": 14, "bit": 0, "mask": 16, "width": 1},
{"i": 62, "op": "mad", "dst": 3, "src": 6, "src2": 0, "imm": "0x41a443a3", "imm2": "0xe69d7919", "rot": 9, "bit": 0, "mask": 16, "width": 1},
{"i": 63, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x2fe0e98b", "imm2": "0xc88e2942", "rot": 5, "bit": 16, "mask": 16, "width": 1}
]
}

View file

@ -0,0 +1,126 @@
#include <metal_stdlib>
using namespace metal;
#define MASK 0x0fffffffu
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
device ulong* out [[buffer(1)]],
constant uint& baseNonce [[buffer(2)]],
device uint* scratch [[buffer(3)]],
constant uint& groups [[buffer(4)]],
constant uint& salt [[buffer(5)]],
uint tid [[thread_position_in_grid]],
uint nthreads [[threads_per_grid]]) {
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
r4 = r0 * r6 + r4; // 3
r7 = r7 ^ dataset[r2 & MASK]; // 4
r4 = r4 ^ dataset[r1 & MASK]; // 5
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
r7 = r7 ^ r5; // 8
r3 = r3 | r4; // 9
r1 = r1 | r2; // 10
r4 = r4 ^ dataset[r3 & MASK]; // 11
r6 = r6 | r2; // 12
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
r5 = r5 ^ r7; // 21
r2 = mulhi(r2, r5); // 22
r3 = r3 ^ dataset[r7 & MASK]; // 23
r7 = mulhi(r7, r3); // 24
r5 = r5 | r4; // 25
r4 = r5 * r2 + r4; // 26
r5 = r5 * r1; // 27
r6 = mulhi(r6, r7); // 28
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
r6 = rotr_var(r6, r7); // 30
r3 = r3 ^ dataset[r1 & MASK]; // 31
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
r4 = r4 ^ dataset[r0 & MASK]; // 37
r1 = r3 * r5 + r1; // 38
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
r2 = rotr_var(r2, r5); // 40
r3 = r3 * r2; // 41
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
r3 = r3 ^ r4; // 43
r3 = r3 ^ dataset[r5 & MASK]; // 44
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
r7 = r7 ^ r1; // 46
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
r7 = mulhi(r7, r5); // 48
r0 = r0 ^ dataset[r2 & MASK]; // 49
r2 = r2 - r6; // 50
r7 = r7 - r5; // 51
r2 = r2 ^ r3; // 52
r7 = r7 - r0; // 53
r3 = r5 * r0 + r3; // 54
r7 = r7 ^ r5; // 55
r2 = r2 ^ dataset[r7 & MASK]; // 56
r5 = r5 - r6; // 57
r1 = r1 ^ dataset[r3 & MASK]; // 58
{ uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
r4 = r4 - r6; // 60
r1 = r1 * r2; // 61
r3 = r6 * r0 + r3; // 62
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}

View file

@ -0,0 +1,128 @@
#include <metal_stdlib>
using namespace metal;
#define MASK 0x0fffffffu
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
device ulong* out [[buffer(1)]],
constant uint& baseNonce [[buffer(2)]],
constant uint* initw [[buffer(3)]],
device uint* scratch [[buffer(4)]],
constant uint& groups [[buffer(5)]],
constant uint& salt [[buffer(6)]],
uint tid [[thread_position_in_grid]],
uint nthreads [[threads_per_grid]]) {
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
r4 = r0 * r6 + r4; // 3
r7 = r7 ^ dataset[r2 & MASK]; // 4
r4 = r4 ^ dataset[r1 & MASK]; // 5
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
r7 = r7 ^ r5; // 8
r3 = r3 | r4; // 9
r1 = r1 | r2; // 10
r4 = r4 ^ dataset[r3 & MASK]; // 11
r6 = r6 | r2; // 12
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
r5 = r5 ^ r7; // 21
r2 = mulhi(r2, r5); // 22
r3 = r3 ^ dataset[r7 & MASK]; // 23
r7 = mulhi(r7, r3); // 24
r5 = r5 | r4; // 25
r4 = r5 * r2 + r4; // 26
r5 = r5 * r1; // 27
r6 = mulhi(r6, r7); // 28
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
r6 = rotr_var(r6, r7); // 30
r3 = r3 ^ dataset[r1 & MASK]; // 31
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
r4 = r4 ^ dataset[r0 & MASK]; // 37
r1 = r3 * r5 + r1; // 38
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
r2 = rotr_var(r2, r5); // 40
r3 = r3 * r2; // 41
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
r3 = r3 ^ r4; // 43
r3 = r3 ^ dataset[r5 & MASK]; // 44
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
r7 = r7 ^ r1; // 46
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
r7 = mulhi(r7, r5); // 48
r0 = r0 ^ dataset[r2 & MASK]; // 49
r2 = r2 - r6; // 50
r7 = r7 - r5; // 51
r2 = r2 ^ r3; // 52
r7 = r7 - r0; // 53
r3 = r5 * r0 + r3; // 54
r7 = r7 ^ r5; // 55
r2 = r2 ^ dataset[r7 & MASK]; // 56
r5 = r5 - r6; // 57
r1 = r1 ^ dataset[r3 & MASK]; // 58
{ uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
r4 = r4 - r6; // 60
r1 = r1 * r2; // 61
r3 = r6 * r0 + r3; // 62
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}

View file

@ -0,0 +1,57 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#define IGNEUM_VEC_WARPS 3
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
{ // base nonce 0
0x62cab4be0ed880e1ull, 0x90842c849268cf52ull, 0x53d4d591f4a12749ull, 0x020429c3d1279eddull, 0x7086876f3a9183fbull, 0xc1f954e8065b6d02ull, 0x7550abd24b6ed9edull, 0x7dcc57f17b255fc9ull,
0x01f16667b6ea326dull, 0x05907ad2b28423a5ull, 0x27e8aad8889ea702ull, 0xe93e61d2ff565fabull, 0x21aa77db95a8790full, 0xdc361b8011f7a395ull, 0xbabf590a4cf6dd58ull, 0x7a9b7eb46cf2b88dull,
0xeb43f11d39490205ull, 0x3c2e3b2ed2b0c6fbull, 0x2bc2293577914bc8ull, 0x66cc059287de6cc0ull, 0x9a2d3a0f30169b20ull, 0xa5756e5027ec459full, 0x734a7fb8546a08b5ull, 0xbc59502ef67511d7ull,
0x869d40616fa13209ull, 0x7e2713e23d7c3e06ull, 0x640f81856492a8d3ull, 0x1f7ce7b42962bc41ull, 0xc474fdd339993867ull, 0xb08d199c21d08397ull, 0x6b021b07d5dd9fdfull, 0x574adb548a4f3be8ull
},
{ // base nonce 4096
0x2956e7703c1553fcull, 0xe06e9c5dd64f0cffull, 0x41d967b788797c3eull, 0xc9beec7571ab7808ull, 0x0d0d99d51ac72942ull, 0xac60f79ae1bb46d6ull, 0xd9b9a509bc33d145ull, 0x512d444977d257f5ull,
0x0e9759a10d773b28ull, 0xb270e841265b2b3dull, 0x90b97771870e53dcull, 0xf8b0bacb1ead0c1bull, 0x32165ca85108736cull, 0x5a907f0cb371d6d1ull, 0x89d4ccf9b6323847ull, 0x495e23339db371dbull,
0xa50d559aa7911894ull, 0xbe561e2c2e64f0ffull, 0x862b141ae3b898eaull, 0x69b52c3068f0544aull, 0x2769f3b4051f9e80ull, 0xb679a28f140a5ccaull, 0x037d194732dce935ull, 0xec9f1e85406dbee9ull,
0x65a0d3f10857795dull, 0x5da0b4908b5cda66ull, 0x1cbdf4f47dad39e4ull, 0x5472317d40d55545ull, 0x24ec2fb5eeff7691ull, 0x4c56104e2454b9beull, 0x8b896d9e85dbf491ull, 0xe80882d5975e09ecull
},
{ // base nonce 1000000
0xa417c0e0494f5f0dull, 0x44cfa8bbf55cb40bull, 0xee534b970664a105ull, 0x2814b1857db92d67ull, 0xd358f7e47b35ff5full, 0xa08faa47e58221c3ull, 0xbb76559b9a4a447bull, 0xd438a3fd1fa2976eull,
0xa08d0e2c88abe900ull, 0x58b3c3ab097c416dull, 0x705de177cf28ccdcull, 0x263f35e27d8cf3aaull, 0xa2c304ccfeb9ae9bull, 0x470ba6ea4e8f661aull, 0xa59e5f33cd8613d9ull, 0xdb887848353dc91cull,
0xd948d1c36b6a98e1ull, 0xd78806062c54882aull, 0x0744e029194938aaull, 0x42b613ec3d9074c6ull, 0x88c75044753d1496ull, 0x940239d09cbd80e1ull, 0x4534dd536f53adf5ull, 0x6f44b9bd4d6564deull,
0xb6d8142c422857d5ull, 0x3b61f0c20b8fb04bull, 0x17eaf3a88b49c9fdull, 0xec536015d760eb1eull, 0xf37e9ad57045cc05ull, 0x588b808bc29ab6caull, 0x7cfa7ae3c6e483c2ull, 0x71e3cc45c07b530cull
}
};
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
static const uint32_t IGNEUM_DS_HEAD[16] = {
0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu,
0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du
};
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u;
// 64 sampled dataset words (index, value) computed on the Mac.
#define IGNEUM_DS_SAMPLES 64
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
};
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u
};
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u,
0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u
};
static const uint32_t IGNEUM_CACHE_LAST[16] = {
0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du,
0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu
};
static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull;

View file

@ -0,0 +1,36 @@
{
"seed": "igneum-genesis",
"day": "2026-10-03",
"dataset_mode": "memory-hard",
"dataset_log2_words": 28,
"mask": "0x0fffffff",
"lanes": 32,
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
"warps": [
{"base_nonce": 0, "expected": [
"0x62cab4be0ed880e1", "0x90842c849268cf52", "0x53d4d591f4a12749", "0x020429c3d1279edd", "0x7086876f3a9183fb", "0xc1f954e8065b6d02", "0x7550abd24b6ed9ed", "0x7dcc57f17b255fc9",
"0x01f16667b6ea326d", "0x05907ad2b28423a5", "0x27e8aad8889ea702", "0xe93e61d2ff565fab", "0x21aa77db95a8790f", "0xdc361b8011f7a395", "0xbabf590a4cf6dd58", "0x7a9b7eb46cf2b88d",
"0xeb43f11d39490205", "0x3c2e3b2ed2b0c6fb", "0x2bc2293577914bc8", "0x66cc059287de6cc0", "0x9a2d3a0f30169b20", "0xa5756e5027ec459f", "0x734a7fb8546a08b5", "0xbc59502ef67511d7",
"0x869d40616fa13209", "0x7e2713e23d7c3e06", "0x640f81856492a8d3", "0x1f7ce7b42962bc41", "0xc474fdd339993867", "0xb08d199c21d08397", "0x6b021b07d5dd9fdf", "0x574adb548a4f3be8"
]},
{"base_nonce": 4096, "expected": [
"0x2956e7703c1553fc", "0xe06e9c5dd64f0cff", "0x41d967b788797c3e", "0xc9beec7571ab7808", "0x0d0d99d51ac72942", "0xac60f79ae1bb46d6", "0xd9b9a509bc33d145", "0x512d444977d257f5",
"0x0e9759a10d773b28", "0xb270e841265b2b3d", "0x90b97771870e53dc", "0xf8b0bacb1ead0c1b", "0x32165ca85108736c", "0x5a907f0cb371d6d1", "0x89d4ccf9b6323847", "0x495e23339db371db",
"0xa50d559aa7911894", "0xbe561e2c2e64f0ff", "0x862b141ae3b898ea", "0x69b52c3068f0544a", "0x2769f3b4051f9e80", "0xb679a28f140a5cca", "0x037d194732dce935", "0xec9f1e85406dbee9",
"0x65a0d3f10857795d", "0x5da0b4908b5cda66", "0x1cbdf4f47dad39e4", "0x5472317d40d55545", "0x24ec2fb5eeff7691", "0x4c56104e2454b9be", "0x8b896d9e85dbf491", "0xe80882d5975e09ec"
]},
{"base_nonce": 1000000, "expected": [
"0xa417c0e0494f5f0d", "0x44cfa8bbf55cb40b", "0xee534b970664a105", "0x2814b1857db92d67", "0xd358f7e47b35ff5f", "0xa08faa47e58221c3", "0xbb76559b9a4a447b", "0xd438a3fd1fa2976e",
"0xa08d0e2c88abe900", "0x58b3c3ab097c416d", "0x705de177cf28ccdc", "0x263f35e27d8cf3aa", "0xa2c304ccfeb9ae9b", "0x470ba6ea4e8f661a", "0xa59e5f33cd8613d9", "0xdb887848353dc91c",
"0xd948d1c36b6a98e1", "0xd78806062c54882a", "0x0744e029194938aa", "0x42b613ec3d9074c6", "0x88c75044753d1496", "0x940239d09cbd80e1", "0x4534dd536f53adf5", "0x6f44b9bd4d6564de",
"0xb6d8142c422857d5", "0x3b61f0c20b8fb04b", "0x17eaf3a88b49c9fd", "0xec536015d760eb1e", "0xf37e9ad57045cc05", "0x588b808bc29ab6ca", "0x7cfa7ae3c6e483c2", "0x71e3cc45c07b530c"
]}
],
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],
"dataset_last_index": 268435455,
"dataset_last": "0xa33ada72",
"dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}],
"cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"],
"cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"],
"cache_fnv1a64": "0x48c4f5bf24166b2e"
}

View file

@ -0,0 +1,291 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
#ifndef IGNEUM_GROUP
#define IGNEUM_GROUP 32
#endif
#ifndef IGNEUM_EXCHANGE
#define IGNEUM_EXCHANGE 0
#endif
#ifdef __OPENCL_VERSION__
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
#if IGNEUM_EXCHANGE == 1
#ifdef cl_khr_subgroups
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
#endif
#ifdef cl_khr_subgroup_shuffle
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
#endif
#elif IGNEUM_EXCHANGE == 2
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
#endif
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
#else
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
#include "emu_opencl.h"
#endif
#if IGNEUM_EXCHANGE == 1
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#elif IGNEUM_EXCHANGE == 2
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#else
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
#endif
static inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
static inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
static inline void mh_chacha_block(const uint* x, uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
static inline void mh_cache_segment(__global uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
static inline void mh_mixer(uint* s, uint rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
// The same constants as memhard.h in this pack (one emitter, three dialects).
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
uint seg = (uint)get_global_id(0);
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
uint t = (uint)get_global_id(0);
if (t < nItems) {
uint s[16];
mh_item(cache, t, s);
__global uint* d = ds + ((ulong)t * 16u);
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
}
}
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
{ uint s_ = r7 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = mul_hi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint s_ = r5 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = mul_hi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}
#if IGNEUM_EXCHANGE != 0
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
}
#endif

View file

@ -0,0 +1,177 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
#include "memhard.h"
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
uint32_t x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
if (t < nItems) {
uint32_t s[16];
mh_item(cache, t, s);
uint32_t* d = ds + (size_t)t * 16u;
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
}
}
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, uint32_t* scratch, uint32_t groups, uint32_t salt) {
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
uint32_t tag = salt + g_;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
{ uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint32_t it = 0u; it < 8u; ++it) {
uint32_t sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
r4 = r4 ^ ds[r1 & mask]; // 5 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = __umulhi(r2, r5); // 22 mulhi
{ uint32_t s_ = r7 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
r7 = __umulhi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = __umulhi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint32_t s_ = r5 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = __umulhi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
}
// Host-side launch wrappers. Declared in program.h, called from host.cu.
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
if (nSegments == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nSegments + block - 1u) / block;
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
return cudaGetLastError();
}
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
if (nItems == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nItems + block - 1u) / block;
igneum_build<<<grid, block>>>(ds, cache, nItems);
return cudaGetLastError();
}
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
igneum_hash<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, scratch, nonces / 32u, salt);
return cudaGetLastError();
}
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
}

View file

@ -0,0 +1,393 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
#ifndef IGNEUM_GROUP
#define IGNEUM_GROUP 32
#endif
#ifndef IGNEUM_EXCHANGE
#define IGNEUM_EXCHANGE 0
#endif
#ifdef __OPENCL_VERSION__
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
#if IGNEUM_EXCHANGE == 1
#ifdef cl_khr_subgroups
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
#endif
#ifdef cl_khr_subgroup_shuffle
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
#endif
#elif IGNEUM_EXCHANGE == 2
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
#endif
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
#else
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
#include "emu_opencl.h"
#endif
#if IGNEUM_EXCHANGE == 1
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#elif IGNEUM_EXCHANGE == 2
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#else
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
#endif
static inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
static inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
static inline void mh_chacha_block(const uint* x, uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
static inline void mh_cache_segment(__global uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
static inline void mh_mixer(uint* s, uint rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
// The same constants as memhard.h in this pack (one emitter, three dialects).
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
uint seg = (uint)get_global_id(0);
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
uint t = (uint)get_global_id(0);
if (t < nItems) {
uint s[16];
mh_item(cache, t, s);
__global uint* d = ds + ((ulong)t * 16u);
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
}
}
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
{ uint s_ = r7 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = mul_hi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint s_ = r5 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = mul_hi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}
#if IGNEUM_EXCHANGE != 0
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
}
#endif
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global uint* scratch, uint groups, uint salt) {
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
{ uint s_ = r7 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = mul_hi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint s_ = r5 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = mul_hi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}

View file

@ -0,0 +1,136 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
// Host declarations (also in program_bound.h if present):
// struct IgneumInitWords { uint32_t w[8]; };
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
struct IgneumInitWords { uint32_t w[8]; };
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, uint32_t* scratch, uint32_t groups, uint32_t salt) {
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
uint32_t tag = salt + g_;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
for (uint32_t it = 0u; it < 8u; ++it) {
uint32_t sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
r4 = r4 ^ ds[r1 & mask]; // 5 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = __umulhi(r2, r5); // 22 mulhi
{ uint32_t s_ = r7 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
r7 = __umulhi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = __umulhi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint32_t s_ = r5 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = __umulhi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
}
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
igneum_hash_bound<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, iw, scratch, nonces / 32u, salt);
return cudaGetLastError();
}
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
}

View file

@ -0,0 +1,108 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#if defined(__CUDACC__)
#define IGNEUM_HD __host__ __device__ __forceinline__
#elif defined(_MSC_VER) && !defined(__cplusplus)
#define IGNEUM_HD static __inline
#else
#define IGNEUM_HD static inline
#endif
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint32_t r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint32_t r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }

View file

@ -0,0 +1,106 @@
#include <metal_stdlib>
using namespace metal;
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
inline void mh_cache_segment(device uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
inline void mh_mixer(thread uint* s, uint rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
// One thread per segment (2^16 threads).
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
mh_cache_segment(cache, gid);
}
// One thread per 64-byte item (dataset words / 16 threads).
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
uint gid [[thread_position_in_grid]]) {
uint s[16];
mh_item(cache, gid, s);
device uint* d = dataset + gid * 16u;
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
}

View file

@ -0,0 +1,67 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#ifndef IGNEUM_NO_CUDA
#include <cuda_runtime.h>
#endif
#define IGNEUM_SEED_STRING "igneum-genesis"
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
#define IGNEUM_GENERATOR 2
#define IGNEUM_PROGRAM_ATTEMPT 0
#define IGNEUM_PROGRAM_ID 0xe0d1dcd155d90573ull
#define IGNEUM_DAY_STRING "2026-10-03"
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
#define IGNEUM_DAY0 0x3067619fu
#define IGNEUM_DAY1 0x3c269176u
#define IGNEUM_DATASET_LOG2 28
#define IGNEUM_MASK 0x0fffffffu
#define IGNEUM_LANES 32
#define IGNEUM_ITERATIONS 8
#define IGNEUM_INSTR_COUNT 64
#define IGNEUM_LOADS_PER_HASH 128
#define IGNEUM_WIDE_LOADS_PER_HASH 0
#define IGNEUM_OP_MIX "add=9 load=8 scratch=8 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 rotl=1"
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
#define IGNEUM_LOAD_CLASS "scr8k128"
#define IGNEUM_LOAD_SLOTS 16
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
#define IGNEUM_LOAD_WIDTH_COUNTS { 8, 0, 0 } // loads of 4, 16, 64 bytes per program
#define IGNEUM_BYTES_PER_HASH 256
#define IGNEUM_FOLD_ROT 11
#define IGNEUM_FOLD_MUL 0x9e3779b1u
// Variant 5: persistent warps, a 128 KiB scratch per launched warp (the host launches N warps and passes scratch,
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
#define IGNEUM_PERSISTENT_WARPS 1
#define IGNEUM_SCRATCH_OPS 8 // scratch read-modify-writes per program (64 per hash)
#define IGNEUM_SCRATCH_SLOTS 256u
#define IGNEUM_SCRATCH_WORDS_PER_LANE 1024u
#define IGNEUM_SCRATCH_BYTES_PER_WARP 131072u
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
#define IGNEUM_DATASET_MODE 1
#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }
#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u }
#define IGNEUM_CACHE_LOG2_WORDS 26
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
#define IGNEUM_CACHE_SEGMENTS 65536u
#define IGNEUM_ITEM_ROUNDS 8
#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u }
#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u }
#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u }
#ifndef IGNEUM_NO_CUDA
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt);
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
#endif

View file

@ -2,7 +2,7 @@
"format": "igneum-program-pack-3",
"generator": 2,
"attempt": 0,
"program_id": "0x2f0992e568f38dc1",
"program_id": "0xe0d1dcd155d90573",
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
"dataset_mode": "memory-hard",
"seed": "igneum-genesis",
@ -15,13 +15,14 @@
"iterations": 8,
"instruction_count": 64,
"loads_per_hash": 128,
"load_class": "scr8",
"load_class": "scr8k128",
"load_slots": 16,
"load_mix_percent_4_16_64": [100, 0, 0],
"load_width_counts_4_16_64": [8, 0, 0],
"bytes_per_hash": 256,
"scratch_ops_per_hash": 64,
"scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
"scratch_kib_per_warp": 128,
"scratch": "variant 5 (measurement only): persistent warps; a 128 KiB scratch per warp of 256 16-byte slots per lane (lane-major); slot = src & 0xff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
"op_mix": {"add": 9, "load": 8, "scratch": 8, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "rotl": 1},
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",

View file

@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -58,7 +58,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
r4 = r0 * r6 + r4; // 3
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
r4 = r4 ^ dataset[r1 & MASK]; // 5
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
@ -70,14 +70,14 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
r5 = r5 ^ r7; // 21
r2 = mulhi(r2, r5); // 22
{ uint s_ = r7 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
{ uint s_ = r7 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
r7 = mulhi(r7, r3); // 24
r5 = r5 | r4; // 25
r4 = r5 * r2 + r4; // 26
@ -86,19 +86,19 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
r6 = rotr_var(r6, r7); // 30
r3 = r3 ^ dataset[r1 & MASK]; // 31
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
r1 = r3 * r5 + r1; // 38
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
r2 = rotr_var(r2, r5); // 40
r3 = r3 * r2; // 41
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
r3 = r3 ^ r4; // 43
{ uint s_ = r5 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
{ uint s_ = r5 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
r7 = r7 ^ r1; // 46
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
@ -113,7 +113,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
r2 = r2 ^ dataset[r7 & MASK]; // 56
r5 = r5 - r6; // 57
r1 = r1 ^ dataset[r3 & MASK]; // 58
{ uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
{ uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
r4 = r4 - r6; // 60
r1 = r1 * r2; // 61
r3 = r6 * r0 + r3; // 62

View file

@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
@ -60,7 +60,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
r4 = r0 * r6 + r4; // 3
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
r4 = r4 ^ dataset[r1 & MASK]; // 5
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
@ -72,14 +72,14 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
r5 = r5 ^ r7; // 21
r2 = mulhi(r2, r5); // 22
{ uint s_ = r7 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
{ uint s_ = r7 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
r7 = mulhi(r7, r3); // 24
r5 = r5 | r4; // 25
r4 = r5 * r2 + r4; // 26
@ -88,19 +88,19 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
r6 = rotr_var(r6, r7); // 30
r3 = r3 ^ dataset[r1 & MASK]; // 31
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
r1 = r3 * r5 + r1; // 38
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
r2 = rotr_var(r2, r5); // 40
r3 = r3 * r2; // 41
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
r3 = r3 ^ r4; // 43
{ uint s_ = r5 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
{ uint s_ = r5 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
r7 = r7 ^ r1; // 46
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
@ -115,7 +115,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
r2 = r2 ^ dataset[r7 & MASK]; // 56
r5 = r5 - r6; // 57
r1 = r1 ^ dataset[r3 & MASK]; // 58
{ uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
{ uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
r4 = r4 - r6; // 60
r1 = r1 * r2; // 61
r3 = r6 * r0 + r3; // 62

View file

@ -0,0 +1,57 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#define IGNEUM_VEC_WARPS 3
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
{ // base nonce 0
0xe814771d17db365cull, 0xc4d4a4ae8b6033caull, 0xa4ea884e5e961196ull, 0x5a4f2e353f13502bull, 0xb5bfae78c5048e4cull, 0x75ed7ff2f8ccb9c1ull, 0xc9d37d26079b1916ull, 0xef3c29ccb9d46163ull,
0xb4cae00d3e73ae8eull, 0x88bc44f26a90e913ull, 0xeb91c51e87da69f8ull, 0x2ee04eeac0ba97b2ull, 0x43e33706056bb735ull, 0x88acef8db41e6bbfull, 0x87295bf633750804ull, 0x4a8310fa3c5393f3ull,
0x9c33366c7aa6a5a6ull, 0x79ba6d5674f78e5cull, 0x03168e07fb7ae416ull, 0xdd45a1b5f54270aeull, 0x0d0084aa74f95b51ull, 0xa04060d3f711930bull, 0xa080e7988516297full, 0xcefa3e2e8de2806full,
0x95dc7a55e10c4010ull, 0x62e809b8cef37be6ull, 0x0366273048e795cfull, 0xc2b04c7deacb1dffull, 0xe83446f7db3a4686ull, 0xe6c7e6575c34651full, 0xee15067b419606d6ull, 0xfe67f3ae51faade1ull
},
{ // base nonce 4096
0xdde6034f4b5824c9ull, 0x07b807430ab9effbull, 0xa581d141cb4bc74aull, 0x0a1b4129b3690618ull, 0xb4d10af1daaec58bull, 0xcf0b63a9aa6b8a96ull, 0x07c20bd30e3eb88cull, 0x32ebffcaafa5df9eull,
0xe1b528deb263ddc3ull, 0xaa6d2e1e7f45c995ull, 0x6017aaa938e837cfull, 0x23445a9b8c8e5addull, 0x024ebd232a344f41ull, 0x67aebe3e79435f84ull, 0xa7d0b7522e88814aull, 0x1d4d9633b57ac637ull,
0x77f500325912fcbdull, 0x9f4bc5d04fbb13b9ull, 0xd7e081a23934d582ull, 0x10992aa1c8a93afeull, 0x1596ba0b47520be7ull, 0x344ed3c63b5a75bcull, 0xdb75c50a7a39c7beull, 0x0e1ccc942ec1fad7ull,
0x81c9d97c4d9605c6ull, 0x1e60918f98df7dc9ull, 0x3e60e5b90d94fe34ull, 0xec7163b65fc01cfdull, 0xb0786922940f66e3ull, 0xb1d049c3e24a38e2ull, 0x5a6a7ac9a0c8ed50ull, 0x7bc50eab43e83a01ull
},
{ // base nonce 1000000
0x1ebe406e6227f5e9ull, 0xc9d07c89dd188990ull, 0x3346fbacbe00f719ull, 0x437f4d678259e06dull, 0xd664758bc7508b7cull, 0xa3428dfd2b480593ull, 0xdfbb1aca3c18aeb6ull, 0xe362f4d90b64ab9full,
0xfaeac4630c51e291ull, 0x2b81c4de8eaa689aull, 0x661b54d4d8763782ull, 0x48839cd831ea402bull, 0xa98dbad17e5f3a49ull, 0x35161bdc5dc87db7ull, 0xcc503293dddde770ull, 0x8960f9fbf15a0e47ull,
0x3d391f32613d80d5ull, 0xd3fb60a843c31905ull, 0x7f70cbe3a4be1f1eull, 0xb67d653a57d143c7ull, 0x08a9210687821d3dull, 0x54adfc465596fd4cull, 0x12cf1cdd36c931e4ull, 0x62fd59b6a002a406ull,
0xfef506618454af46ull, 0x8ab9f4c86cfefbe0ull, 0xb54d7b600ee5cbdbull, 0x9400685a172c31ccull, 0x0cdc0ca4f87996dcull, 0x050be1f1631ab65full, 0x79884be8f2e9a1f2ull, 0x824d03dfe16bcb91ull
}
};
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
static const uint32_t IGNEUM_DS_HEAD[16] = {
0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu,
0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du
};
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u;
// 64 sampled dataset words (index, value) computed on the Mac.
#define IGNEUM_DS_SAMPLES 64
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
};
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u
};
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u,
0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u
};
static const uint32_t IGNEUM_CACHE_LAST[16] = {
0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du,
0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu
};
static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull;

View file

@ -0,0 +1,36 @@
{
"seed": "igneum-genesis",
"day": "2026-10-03",
"dataset_mode": "memory-hard",
"dataset_log2_words": 28,
"mask": "0x0fffffff",
"lanes": 32,
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
"warps": [
{"base_nonce": 0, "expected": [
"0xe814771d17db365c", "0xc4d4a4ae8b6033ca", "0xa4ea884e5e961196", "0x5a4f2e353f13502b", "0xb5bfae78c5048e4c", "0x75ed7ff2f8ccb9c1", "0xc9d37d26079b1916", "0xef3c29ccb9d46163",
"0xb4cae00d3e73ae8e", "0x88bc44f26a90e913", "0xeb91c51e87da69f8", "0x2ee04eeac0ba97b2", "0x43e33706056bb735", "0x88acef8db41e6bbf", "0x87295bf633750804", "0x4a8310fa3c5393f3",
"0x9c33366c7aa6a5a6", "0x79ba6d5674f78e5c", "0x03168e07fb7ae416", "0xdd45a1b5f54270ae", "0x0d0084aa74f95b51", "0xa04060d3f711930b", "0xa080e7988516297f", "0xcefa3e2e8de2806f",
"0x95dc7a55e10c4010", "0x62e809b8cef37be6", "0x0366273048e795cf", "0xc2b04c7deacb1dff", "0xe83446f7db3a4686", "0xe6c7e6575c34651f", "0xee15067b419606d6", "0xfe67f3ae51faade1"
]},
{"base_nonce": 4096, "expected": [
"0xdde6034f4b5824c9", "0x07b807430ab9effb", "0xa581d141cb4bc74a", "0x0a1b4129b3690618", "0xb4d10af1daaec58b", "0xcf0b63a9aa6b8a96", "0x07c20bd30e3eb88c", "0x32ebffcaafa5df9e",
"0xe1b528deb263ddc3", "0xaa6d2e1e7f45c995", "0x6017aaa938e837cf", "0x23445a9b8c8e5add", "0x024ebd232a344f41", "0x67aebe3e79435f84", "0xa7d0b7522e88814a", "0x1d4d9633b57ac637",
"0x77f500325912fcbd", "0x9f4bc5d04fbb13b9", "0xd7e081a23934d582", "0x10992aa1c8a93afe", "0x1596ba0b47520be7", "0x344ed3c63b5a75bc", "0xdb75c50a7a39c7be", "0x0e1ccc942ec1fad7",
"0x81c9d97c4d9605c6", "0x1e60918f98df7dc9", "0x3e60e5b90d94fe34", "0xec7163b65fc01cfd", "0xb0786922940f66e3", "0xb1d049c3e24a38e2", "0x5a6a7ac9a0c8ed50", "0x7bc50eab43e83a01"
]},
{"base_nonce": 1000000, "expected": [
"0x1ebe406e6227f5e9", "0xc9d07c89dd188990", "0x3346fbacbe00f719", "0x437f4d678259e06d", "0xd664758bc7508b7c", "0xa3428dfd2b480593", "0xdfbb1aca3c18aeb6", "0xe362f4d90b64ab9f",
"0xfaeac4630c51e291", "0x2b81c4de8eaa689a", "0x661b54d4d8763782", "0x48839cd831ea402b", "0xa98dbad17e5f3a49", "0x35161bdc5dc87db7", "0xcc503293dddde770", "0x8960f9fbf15a0e47",
"0x3d391f32613d80d5", "0xd3fb60a843c31905", "0x7f70cbe3a4be1f1e", "0xb67d653a57d143c7", "0x08a9210687821d3d", "0x54adfc465596fd4c", "0x12cf1cdd36c931e4", "0x62fd59b6a002a406",
"0xfef506618454af46", "0x8ab9f4c86cfefbe0", "0xb54d7b600ee5cbdb", "0x9400685a172c31cc", "0x0cdc0ca4f87996dc", "0x050be1f1631ab65f", "0x79884be8f2e9a1f2", "0x824d03dfe16bcb91"
]}
],
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],
"dataset_last_index": 268435455,
"dataset_last": "0xa33ada72",
"dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}],
"cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"],
"cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"],
"cache_fnv1a64": "0x48c4f5bf24166b2e"
}

View file

@ -0,0 +1,291 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
#ifndef IGNEUM_GROUP
#define IGNEUM_GROUP 32
#endif
#ifndef IGNEUM_EXCHANGE
#define IGNEUM_EXCHANGE 0
#endif
#ifdef __OPENCL_VERSION__
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
#if IGNEUM_EXCHANGE == 1
#ifdef cl_khr_subgroups
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
#endif
#ifdef cl_khr_subgroup_shuffle
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
#endif
#elif IGNEUM_EXCHANGE == 2
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
#endif
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
#else
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
#include "emu_opencl.h"
#endif
#if IGNEUM_EXCHANGE == 1
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#elif IGNEUM_EXCHANGE == 2
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#else
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
#endif
static inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
static inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
static inline void mh_chacha_block(const uint* x, uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
static inline void mh_cache_segment(__global uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
static inline void mh_mixer(uint* s, uint rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
// The same constants as memhard.h in this pack (one emitter, three dialects).
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
uint seg = (uint)get_global_id(0);
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
uint t = (uint)get_global_id(0);
if (t < nItems) {
uint s[16];
mh_item(cache, t, s);
__global uint* d = ds + ((ulong)t * 16u);
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
}
}
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
{ uint s_ = r7 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = mul_hi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint s_ = r5 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = mul_hi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}
#if IGNEUM_EXCHANGE != 0
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
}
#endif

View file

@ -0,0 +1,177 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
#include "memhard.h"
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
uint32_t x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
if (t < nItems) {
uint32_t s[16];
mh_item(cache, t, s);
uint32_t* d = ds + (size_t)t * 16u;
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
}
}
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, uint32_t* scratch, uint32_t groups, uint32_t salt) {
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
uint32_t tag = salt + g_;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
{ uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint32_t it = 0u; it < 8u; ++it) {
uint32_t sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
r4 = r4 ^ ds[r1 & mask]; // 5 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = __umulhi(r2, r5); // 22 mulhi
{ uint32_t s_ = r7 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
r7 = __umulhi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = __umulhi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint32_t s_ = r5 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = __umulhi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
}
// Host-side launch wrappers. Declared in program.h, called from host.cu.
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
if (nSegments == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nSegments + block - 1u) / block;
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
return cudaGetLastError();
}
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
if (nItems == 0u) return cudaErrorInvalidValue;
uint32_t block = 256u;
uint32_t grid = (nItems + block - 1u) / block;
igneum_build<<<grid, block>>>(ds, cache, nItems);
return cudaGetLastError();
}
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
igneum_hash<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, scratch, nonces / 32u, salt);
return cudaGetLastError();
}
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
}

View file

@ -0,0 +1,393 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
#ifndef IGNEUM_GROUP
#define IGNEUM_GROUP 32
#endif
#ifndef IGNEUM_EXCHANGE
#define IGNEUM_EXCHANGE 0
#endif
#ifdef __OPENCL_VERSION__
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
#if IGNEUM_EXCHANGE == 1
#ifdef cl_khr_subgroups
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
#endif
#ifdef cl_khr_subgroup_shuffle
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
#endif
#elif IGNEUM_EXCHANGE == 2
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
#endif
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
#else
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
#include "emu_opencl.h"
#endif
#if IGNEUM_EXCHANGE == 1
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#elif IGNEUM_EXCHANGE == 2
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
#else
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
#endif
static inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
static inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
static inline void mh_chacha_block(const uint* x, uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
static inline void mh_cache_segment(__global uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
static inline void mh_mixer(uint* s, uint rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
// The same constants as memhard.h in this pack (one emitter, three dialects).
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
uint seg = (uint)get_global_id(0);
if (seg < nSegments) mh_cache_segment(cache, seg);
}
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
uint t = (uint)get_global_id(0);
if (t < nItems) {
uint s[16];
mh_item(cache, t, s);
__global uint* d = ds + ((ulong)t * 16u);
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
}
}
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
{ uint s_ = r7 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = mul_hi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint s_ = r5 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = mul_hi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}
#if IGNEUM_EXCHANGE != 0
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
}
#endif
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global uint* scratch, uint groups, uint salt) {
uint lane = (uint)get_global_id(0) & 31u;
uint warp_ = (uint)get_global_id(0) >> 5;
uint nwarps_ = (uint)get_global_size(0) >> 5;
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint lid = (uint)get_local_id(0);
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
#if IGNEUM_EXCHANGE == 0
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
uint xk = 0u;
#else
(void)lid;
#endif
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
r4 = r4 ^ ds[r1 & mask]; // 5 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = mul_hi(r2, r5); // 22 mulhi
{ uint s_ = r7 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
r7 = mul_hi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = mul_hi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint s_ = r5 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = mul_hi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}

View file

@ -0,0 +1,136 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
// Host declarations (also in program_bound.h if present):
// struct IgneumInitWords { uint32_t w[8]; };
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
#include <cuda_runtime.h>
#include <cstdint>
#include "program.h"
struct IgneumInitWords { uint32_t w[8]; };
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, uint32_t* scratch, uint32_t groups, uint32_t salt) {
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
uint32_t gid = g_ * 32u + lane;
uint32_t gbase = baseNonce + g_ * 32u;
uint32_t tag = salt + g_;
uint32_t nonce = baseNonce + gid;
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
for (uint32_t it = 0u; it < 8u; ++it) {
uint32_t sel = r0;
r2 = r3 * r4 + r2; // 0 mad
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
r4 = r0 * r6 + r4; // 3 mad
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
r4 = r4 ^ ds[r1 & mask]; // 5 load
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
r7 = r7 ^ r5; // 8 xor
r3 = r3 | r4; // 9 or
r1 = r1 | r2; // 10 or
r4 = r4 ^ ds[r3 & mask]; // 11 load
r6 = r6 | r2; // 12 or
r2 = r2 * r5; // 13 mul
r1 = r1 ^ ds[r2 & mask]; // 14 load
r7 = rotl_imm(r7, 1u); // 15 rotl
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
r7 = r7 ^ ds[r4 & mask]; // 17 load
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
r4 = r0 * r2 + r4; // 19 mad
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
r5 = r5 ^ r7; // 21 xor
r2 = __umulhi(r2, r5); // 22 mulhi
{ uint32_t s_ = r7 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
r7 = __umulhi(r7, r3); // 24 mulhi
r5 = r5 | r4; // 25 or
r4 = r5 * r2 + r4; // 26 mad
r5 = r5 * r1; // 27 mul
r6 = __umulhi(r6, r7); // 28 mulhi
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
r6 = rotr_var(r6, r7); // 30 rotr
r3 = r3 ^ ds[r1 & mask]; // 31 load
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
r0 = r0 * r3; // 35 mul
r2 = r2 ^ r5; // 36 xor
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
r1 = r3 * r5 + r1; // 38 mad
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
r2 = rotr_var(r2, r5); // 40 rotr
r3 = r3 * r2; // 41 mul
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
r3 = r3 ^ r4; // 43 xor
{ uint32_t s_ = r5 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
r7 = r7 ^ r1; // 46 xor
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
r7 = __umulhi(r7, r5); // 48 mulhi
r0 = r0 ^ ds[r2 & mask]; // 49 load
r2 = r2 - r6; // 50 sub
r7 = r7 - r5; // 51 sub
r2 = r2 ^ r3; // 52 xor
r7 = r7 - r0; // 53 sub
r3 = r5 * r0 + r3; // 54 mad
r7 = r7 ^ r5; // 55 xor
r2 = r2 ^ ds[r7 & mask]; // 56 load
r5 = r5 - r6; // 57 sub
r1 = r1 ^ ds[r3 & mask]; // 58 load
{ uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
r4 = r4 - r6; // 60 sub
r1 = r1 * r2; // 61 mul
r3 = r6 * r0 + r3; // 62 mad
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
}
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
}
}
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
uint32_t block = 32u * blockWarps;
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
igneum_hash_bound<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, iw, scratch, nonces / 32u, salt);
return cudaGetLastError();
}
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
cudaFuncAttributes attr;
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
if (e != cudaSuccess) return e;
*numRegs = attr.numRegs;
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
}

View file

@ -0,0 +1,108 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#if defined(__CUDACC__)
#define IGNEUM_HD __host__ __device__ __forceinline__
#elif defined(_MSC_VER) && !defined(__cplusplus)
#define IGNEUM_HD static __inline
#else
#define IGNEUM_HD static inline
#endif
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint32_t r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint32_t r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }

View file

@ -0,0 +1,106 @@
#include <metal_stdlib>
using namespace metal;
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
#define MH_CACHE_LINE_MASK 0x003fffffu
#define MH_SEGMENT_LINES 64u
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
// y = ChaCha12 core(x) + x
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
for (uint r = 0u; r < 6u; ++r) {
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
}
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
}
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
inline void mh_cache_segment(device uint* cache, uint seg) {
uint prev[16]; uint x[16]; uint y[16];
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
x[4] = 0x3067619fu ^ prev[4];
x[5] = 0x3c269176u ^ prev[5];
x[6] = 0x84a03b03u ^ prev[6];
x[7] = 0xf8c63294u ^ prev[7];
x[8] = 0xff977c5bu ^ prev[8];
x[9] = 0xe60def3eu ^ prev[9];
x[10] = 0x63630141u ^ prev[10];
x[11] = 0xb8fbcb58u ^ prev[11];
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
mh_chacha_block(x, y);
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
}
}
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
inline void mh_mixer(thread uint* s, uint rk) {
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
}
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
s[0] = 0x3067619fu;
s[1] = 0x3c269176u;
s[2] = 0x84a03b03u;
s[3] = 0xf8c63294u;
s[4] = 0xff977c5bu;
s[5] = 0xe60def3eu;
s[6] = 0x63630141u;
s[7] = 0xb8fbcb58u;
s[8] = t * 0x42146205u + 0xbab68293u;
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
s[11] = t * 0x6728907fu + 0xe62b8997u;
s[12] = t * 0xd81d9751u + 0xc9c80297u;
s[13] = t * 0x132952c3u + 0xf74a1654u;
s[14] = t * 0xf60de277u + 0x3d704af5u;
s[15] = t * 0x05358035u + 0x3cf522b7u;
for (uint r = 0u; r < 8u; ++r) {
mh_mixer(s, 0x9E3779B9u * (r + 1u));
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
}
mh_mixer(s, 0x9E3779B9u * 9u);
}
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
// One thread per segment (2^16 threads).
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
mh_cache_segment(cache, gid);
}
// One thread per 64-byte item (dataset words / 16 threads).
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
uint gid [[thread_position_in_grid]]) {
uint s[16];
mh_item(cache, gid, s);
device uint* d = dataset + gid * 16u;
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
}

View file

@ -15,7 +15,7 @@
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
#define IGNEUM_GENERATOR 2
#define IGNEUM_PROGRAM_ATTEMPT 0
#define IGNEUM_PROGRAM_ID 0x2f0992e568f38dc1ull
#define IGNEUM_PROGRAM_ID 0xe0d27cd155da1553ull
#define IGNEUM_DAY_STRING "2026-10-03"
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
#define IGNEUM_DAY0 0x3067619fu
@ -30,20 +30,20 @@
#define IGNEUM_OP_MIX "add=9 load=8 scratch=8 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 rotl=1"
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
#define IGNEUM_LOAD_CLASS "scr8"
#define IGNEUM_LOAD_CLASS "scr8k32"
#define IGNEUM_LOAD_SLOTS 16
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
#define IGNEUM_LOAD_WIDTH_COUNTS { 8, 0, 0 } // loads of 4, 16, 64 bytes per program
#define IGNEUM_BYTES_PER_HASH 256
#define IGNEUM_FOLD_ROT 11
#define IGNEUM_FOLD_MUL 0x9e3779b1u
// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch,
// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch,
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
#define IGNEUM_PERSISTENT_WARPS 1
#define IGNEUM_SCRATCH_OPS 8 // scratch read-modify-writes per program (64 per hash)
#define IGNEUM_SCRATCH_SLOTS 2048u
#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u
#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u
#define IGNEUM_SCRATCH_SLOTS 64u
#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u
#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
#define IGNEUM_DATASET_MODE 1

View file

@ -0,0 +1,130 @@
{
"format": "igneum-program-pack-3",
"generator": 2,
"attempt": 0,
"program_id": "0xe0d27cd155da1553",
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
"dataset_mode": "memory-hard",
"seed": "igneum-genesis",
"seed_bytes": "69676e65756d2d67656e65736973",
"seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"],
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
"lanes": 32,
"registers": 8,
"iterations": 8,
"instruction_count": 64,
"loads_per_hash": 128,
"load_class": "scr8k32",
"load_slots": 16,
"load_mix_percent_4_16_64": [100, 0, 0],
"load_width_counts_4_16_64": [8, 0, 0],
"bytes_per_hash": 256,
"scratch_ops_per_hash": 64,
"scratch_kib_per_warp": 32,
"scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
"op_mix": {"add": 9, "load": 8, "scratch": 8, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "rotl": 1},
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
"op_semantics": {
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
"sub": "dst = dst - src",
"mul": "dst = dst * src (low 32)",
"mulhi": "dst = high 32 bits of dst * src",
"xor": "dst = dst ^ src",
"or": "dst = dst | src",
"rotl": "dst = rotl(dst, rot), rot in 1..31",
"rotr": "dst = rotr(dst, src & 31)",
"mad": "dst = src * src2 + dst",
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
"load": "dst = dst ^ dataset[src & dataset.mask]",
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
},
"dataset": {
"log2_words": 28,
"bytes": 1073741824,
"mask": "0x0fffffff",
"day": "2026-10-03",
"day_bytes": "6461792f323032362d31302d3033",
"day_words_from": "seed_words_from_bytes(day_bytes)",
"d0": "0x3067619f",
"d1": "0x3c269176",
"mode": "memory-hard",
"spec": "proto-metal/MEMHARD.md",
"key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"],
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s",
"word": "dataset[w] = item(w >> 4)[w & 15]"
},
"instructions": [
{"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1},
{"i": 1, "op": "add", "dst": 1, "src": 7, "src2": 2, "imm": "0x42da7657", "imm2": "0xc3bd2355", "rot": 25, "bit": 4, "mask": 16, "width": 1},
{"i": 2, "op": "add", "dst": 2, "src": 3, "src2": 0, "imm": "0x61f0b51c", "imm2": "0x2735a174", "rot": 4, "bit": 26, "mask": 2, "width": 1},
{"i": 3, "op": "mad", "dst": 4, "src": 0, "src2": 6, "imm": "0x679648a8", "imm2": "0x3044ba32", "rot": 31, "bit": 31, "mask": 4, "width": 1},
{"i": 4, "op": "scratch", "dst": 7, "src": 2, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1},
{"i": 5, "op": "load", "dst": 4, "src": 1, "src2": 2, "imm": "0x987c017a", "imm2": "0xf4d60559", "rot": 2, "bit": 0, "mask": 4, "width": 1},
{"i": 6, "op": "shfl", "dst": 6, "src": 3, "src2": 7, "imm": "0x6ea7b2df", "imm2": "0x9fce5071", "rot": 7, "bit": 15, "mask": 4, "width": 1},
{"i": 7, "op": "shfl", "dst": 1, "src": 5, "src2": 1, "imm": "0x26a2ecde", "imm2": "0xfec6ad22", "rot": 15, "bit": 11, "mask": 8, "width": 1},
{"i": 8, "op": "xor", "dst": 7, "src": 5, "src2": 2, "imm": "0xbe4b445c", "imm2": "0x17a5a9c7", "rot": 8, "bit": 8, "mask": 1, "width": 1},
{"i": 9, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1},
{"i": 10, "op": "or", "dst": 1, "src": 2, "src2": 3, "imm": "0x4e7dc10d", "imm2": "0x196d165c", "rot": 14, "bit": 27, "mask": 16, "width": 1},
{"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 1, "imm": "0xc5c3b55d", "imm2": "0xec061424", "rot": 26, "bit": 27, "mask": 8, "width": 1},
{"i": 12, "op": "or", "dst": 6, "src": 2, "src2": 3, "imm": "0x306542fe", "imm2": "0x1bb1b429", "rot": 31, "bit": 0, "mask": 2, "width": 1},
{"i": 13, "op": "mul", "dst": 2, "src": 5, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1},
{"i": 14, "op": "load", "dst": 1, "src": 2, "src2": 5, "imm": "0x028b4d37", "imm2": "0x7bbd78ea", "rot": 15, "bit": 2, "mask": 8, "width": 1},
{"i": 15, "op": "rotl", "dst": 7, "src": 6, "src2": 6, "imm": "0x5c88a1a7", "imm2": "0x5c628769", "rot": 1, "bit": 3, "mask": 8, "width": 1},
{"i": 16, "op": "scratch", "dst": 3, "src": 6, "src2": 7, "imm": "0xbac2ae81", "imm2": "0xcbbc7bdb", "rot": 18, "bit": 8, "mask": 8, "width": 1},
{"i": 17, "op": "load", "dst": 7, "src": 4, "src2": 2, "imm": "0xe8ab93e9", "imm2": "0xa00de107", "rot": 2, "bit": 1, "mask": 16, "width": 1},
{"i": 18, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1},
{"i": 19, "op": "mad", "dst": 4, "src": 0, "src2": 2, "imm": "0x5fba7bc2", "imm2": "0xdf099cfb", "rot": 4, "bit": 15, "mask": 16, "width": 1},
{"i": 20, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0x0a3056de", "imm2": "0x7f0c25c3", "rot": 27, "bit": 13, "mask": 8, "width": 1},
{"i": 21, "op": "xor", "dst": 5, "src": 7, "src2": 4, "imm": "0xbd066e1d", "imm2": "0x6d3ddc5a", "rot": 2, "bit": 29, "mask": 1, "width": 1},
{"i": 22, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1},
{"i": 23, "op": "scratch", "dst": 3, "src": 7, "src2": 2, "imm": "0xc7fcfc8f", "imm2": "0x8528b94f", "rot": 17, "bit": 13, "mask": 4, "width": 1},
{"i": 24, "op": "mulhi", "dst": 7, "src": 3, "src2": 5, "imm": "0xd91641e8", "imm2": "0xaf77faf2", "rot": 22, "bit": 21, "mask": 1, "width": 1},
{"i": 25, "op": "or", "dst": 5, "src": 4, "src2": 0, "imm": "0x84c03868", "imm2": "0xf6c691b7", "rot": 29, "bit": 14, "mask": 8, "width": 1},
{"i": 26, "op": "mad", "dst": 4, "src": 5, "src2": 2, "imm": "0x3bb2b6ba", "imm2": "0x49d95fd5", "rot": 1, "bit": 5, "mask": 8, "width": 1},
{"i": 27, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1},
{"i": 28, "op": "mulhi", "dst": 6, "src": 7, "src2": 6, "imm": "0xd69c4715", "imm2": "0xe0ebc4ce", "rot": 29, "bit": 2, "mask": 8, "width": 1},
{"i": 29, "op": "add", "dst": 6, "src": 1, "src2": 2, "imm": "0x3b2d2124", "imm2": "0x187a9128", "rot": 1, "bit": 9, "mask": 16, "width": 1},
{"i": 30, "op": "rotr", "dst": 6, "src": 7, "src2": 0, "imm": "0x5c64a589", "imm2": "0x61c9a38d", "rot": 17, "bit": 21, "mask": 16, "width": 1},
{"i": 31, "op": "load", "dst": 3, "src": 1, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1},
{"i": 32, "op": "scratch", "dst": 1, "src": 0, "src2": 7, "imm": "0xcc7972c4", "imm2": "0xad098d15", "rot": 30, "bit": 21, "mask": 8, "width": 1},
{"i": 33, "op": "add", "dst": 0, "src": 4, "src2": 4, "imm": "0x2c35699f", "imm2": "0x351dde38", "rot": 21, "bit": 18, "mask": 4, "width": 1},
{"i": 34, "op": "scratch", "dst": 0, "src": 2, "src2": 3, "imm": "0xfae8902b", "imm2": "0x5cd8306f", "rot": 5, "bit": 28, "mask": 16, "width": 1},
{"i": 35, "op": "mul", "dst": 0, "src": 3, "src2": 1, "imm": "0x4fa3f3db", "imm2": "0xdbf37e75", "rot": 7, "bit": 18, "mask": 4, "width": 1},
{"i": 36, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1},
{"i": 37, "op": "scratch", "dst": 4, "src": 0, "src2": 0, "imm": "0x04cc1d55", "imm2": "0x35c52d04", "rot": 11, "bit": 14, "mask": 2, "width": 1},
{"i": 38, "op": "mad", "dst": 1, "src": 3, "src2": 5, "imm": "0x3958f280", "imm2": "0x8713c7e1", "rot": 5, "bit": 23, "mask": 16, "width": 1},
{"i": 39, "op": "add", "dst": 0, "src": 3, "src2": 3, "imm": "0xa907b90b", "imm2": "0x1b053acf", "rot": 30, "bit": 25, "mask": 16, "width": 1},
{"i": 40, "op": "rotr", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1},
{"i": 41, "op": "mul", "dst": 3, "src": 2, "src2": 4, "imm": "0x49087d74", "imm2": "0x6348b489", "rot": 17, "bit": 9, "mask": 16, "width": 1},
{"i": 42, "op": "add", "dst": 1, "src": 5, "src2": 1, "imm": "0xa32e000c", "imm2": "0x6058c2e3", "rot": 25, "bit": 20, "mask": 8, "width": 1},
{"i": 43, "op": "xor", "dst": 3, "src": 4, "src2": 2, "imm": "0x3dad0eb6", "imm2": "0xb97578cb", "rot": 3, "bit": 27, "mask": 1, "width": 1},
{"i": 44, "op": "scratch", "dst": 3, "src": 5, "src2": 7, "imm": "0x374aec92", "imm2": "0x626f11df", "rot": 20, "bit": 18, "mask": 8, "width": 1},
{"i": 45, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1},
{"i": 46, "op": "xor", "dst": 7, "src": 1, "src2": 0, "imm": "0xef6ac348", "imm2": "0x963bb7e6", "rot": 26, "bit": 3, "mask": 8, "width": 1},
{"i": 47, "op": "add", "dst": 0, "src": 3, "src2": 0, "imm": "0x838b5065", "imm2": "0x36360066", "rot": 3, "bit": 31, "mask": 4, "width": 1},
{"i": 48, "op": "mulhi", "dst": 7, "src": 5, "src2": 0, "imm": "0x8458f7ac", "imm2": "0xc1c15026", "rot": 27, "bit": 15, "mask": 8, "width": 1},
{"i": 49, "op": "load", "dst": 0, "src": 2, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1},
{"i": 50, "op": "sub", "dst": 2, "src": 6, "src2": 0, "imm": "0x2baec8c9", "imm2": "0x4390f156", "rot": 3, "bit": 12, "mask": 8, "width": 1},
{"i": 51, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x19234061", "imm2": "0xe84dfade", "rot": 4, "bit": 19, "mask": 1, "width": 1},
{"i": 52, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0xdc2cd71e", "imm2": "0x1b5d334b", "rot": 9, "bit": 8, "mask": 8, "width": 1},
{"i": 53, "op": "sub", "dst": 7, "src": 0, "src2": 4, "imm": "0x605c31ec", "imm2": "0x9923ff88", "rot": 28, "bit": 25, "mask": 4, "width": 1},
{"i": 54, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1},
{"i": 55, "op": "xor", "dst": 7, "src": 5, "src2": 5, "imm": "0xad7493e7", "imm2": "0x3e400372", "rot": 13, "bit": 8, "mask": 1, "width": 1},
{"i": 56, "op": "load", "dst": 2, "src": 7, "src2": 1, "imm": "0x87e933c9", "imm2": "0x8c854c1b", "rot": 17, "bit": 3, "mask": 8, "width": 1},
{"i": 57, "op": "sub", "dst": 5, "src": 6, "src2": 5, "imm": "0x11be3bc9", "imm2": "0xbbaa8e24", "rot": 6, "bit": 5, "mask": 16, "width": 1},
{"i": 58, "op": "load", "dst": 1, "src": 3, "src2": 2, "imm": "0xa732351a", "imm2": "0xc01349cd", "rot": 14, "bit": 17, "mask": 16, "width": 1},
{"i": 59, "op": "scratch", "dst": 1, "src": 4, "src2": 0, "imm": "0xb20547b2", "imm2": "0xc94655de", "rot": 27, "bit": 30, "mask": 1, "width": 1},
{"i": 60, "op": "sub", "dst": 4, "src": 6, "src2": 7, "imm": "0x67cf904c", "imm2": "0x6873b216", "rot": 27, "bit": 7, "mask": 16, "width": 1},
{"i": 61, "op": "mul", "dst": 1, "src": 2, "src2": 7, "imm": "0x93ab0bf4", "imm2": "0x96158375", "rot": 14, "bit": 0, "mask": 16, "width": 1},
{"i": 62, "op": "mad", "dst": 3, "src": 6, "src2": 0, "imm": "0x41a443a3", "imm2": "0xe69d7919", "rot": 9, "bit": 0, "mask": 16, "width": 1},
{"i": 63, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x2fe0e98b", "imm2": "0xc88e2942", "rot": 5, "bit": 16, "mask": 16, "width": 1}
]
}

View file

@ -0,0 +1,126 @@
#include <metal_stdlib>
using namespace metal;
#define MASK 0x0fffffffu
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
device ulong* out [[buffer(1)]],
constant uint& baseNonce [[buffer(2)]],
device uint* scratch [[buffer(3)]],
constant uint& groups [[buffer(4)]],
constant uint& salt [[buffer(5)]],
uint tid [[thread_position_in_grid]],
uint nthreads [[threads_per_grid]]) {
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
r4 = r0 * r6 + r4; // 3
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
r4 = r4 ^ dataset[r1 & MASK]; // 5
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
r7 = r7 ^ r5; // 8
r3 = r3 | r4; // 9
r1 = r1 | r2; // 10
r4 = r4 ^ dataset[r3 & MASK]; // 11
r6 = r6 | r2; // 12
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
r5 = r5 ^ r7; // 21
r2 = mulhi(r2, r5); // 22
{ uint s_ = r7 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
r7 = mulhi(r7, r3); // 24
r5 = r5 | r4; // 25
r4 = r5 * r2 + r4; // 26
r5 = r5 * r1; // 27
r6 = mulhi(r6, r7); // 28
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
r6 = rotr_var(r6, r7); // 30
r3 = r3 ^ dataset[r1 & MASK]; // 31
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
r1 = r3 * r5 + r1; // 38
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
r2 = rotr_var(r2, r5); // 40
r3 = r3 * r2; // 41
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
r3 = r3 ^ r4; // 43
{ uint s_ = r5 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
r7 = r7 ^ r1; // 46
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
r7 = mulhi(r7, r5); // 48
r0 = r0 ^ dataset[r2 & MASK]; // 49
r2 = r2 - r6; // 50
r7 = r7 - r5; // 51
r2 = r2 ^ r3; // 52
r7 = r7 - r0; // 53
r3 = r5 * r0 + r3; // 54
r7 = r7 ^ r5; // 55
r2 = r2 ^ dataset[r7 & MASK]; // 56
r5 = r5 - r6; // 57
r1 = r1 ^ dataset[r3 & MASK]; // 58
{ uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
r4 = r4 - r6; // 60
r1 = r1 * r2; // 61
r3 = r6 * r0 + r3; // 62
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}

View file

@ -0,0 +1,128 @@
#include <metal_stdlib>
using namespace metal;
#define MASK 0x0fffffffu
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
inline uint splitmix32(uint x) {
x ^= x >> 16; x *= 0x7feb352du;
x ^= x >> 15; x *= 0x846ca68bu;
x ^= x >> 16;
return x;
}
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
inline uint ds_elem(uint i, uint d0, uint d1) {
uint x = i ^ d0;
x *= 0x9E3779B1u; x ^= x >> 15;
x += d1;
x *= 0x85EBCA77u; x ^= x >> 13;
x *= 0xC2B2AE3Du; x ^= x >> 16;
return x;
}
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
device ulong* out [[buffer(1)]],
constant uint& baseNonce [[buffer(2)]],
constant uint* initw [[buffer(3)]],
device uint* scratch [[buffer(4)]],
constant uint& groups [[buffer(5)]],
constant uint& salt [[buffer(6)]],
uint tid [[thread_position_in_grid]],
uint nthreads [[threads_per_grid]]) {
uint lane = tid & 31u;
uint warp_ = tid >> 5;
uint nwarps_ = nthreads >> 5;
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
uint gid = g_ * 32u + lane;
uint gbase = baseNonce + g_ * 32u;
uint tag = salt + g_;
uint nonce = baseNonce + gid;
uint r0, r1, r2, r3, r4, r5, r6, r7;
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
for (uint it = 0u; it < 8u; ++it) {
uint sel = r0;
r2 = r3 * r4 + r2; // 0
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
r4 = r0 * r6 + r4; // 3
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
r4 = r4 ^ dataset[r1 & MASK]; // 5
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
r7 = r7 ^ r5; // 8
r3 = r3 | r4; // 9
r1 = r1 | r2; // 10
r4 = r4 ^ dataset[r3 & MASK]; // 11
r6 = r6 | r2; // 12
r2 = r2 * r5; // 13
r1 = r1 ^ dataset[r2 & MASK]; // 14
r7 = rotl_imm(r7, 1u); // 15
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
r7 = r7 ^ dataset[r4 & MASK]; // 17
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
r4 = r0 * r2 + r4; // 19
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
r5 = r5 ^ r7; // 21
r2 = mulhi(r2, r5); // 22
{ uint s_ = r7 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
r7 = mulhi(r7, r3); // 24
r5 = r5 | r4; // 25
r4 = r5 * r2 + r4; // 26
r5 = r5 * r1; // 27
r6 = mulhi(r6, r7); // 28
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
r6 = rotr_var(r6, r7); // 30
r3 = r3 ^ dataset[r1 & MASK]; // 31
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
r0 = r0 * r3; // 35
r2 = r2 ^ r5; // 36
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
r1 = r3 * r5 + r1; // 38
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
r2 = rotr_var(r2, r5); // 40
r3 = r3 * r2; // 41
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
r3 = r3 ^ r4; // 43
{ uint s_ = r5 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
r7 = r7 ^ r1; // 46
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
r7 = mulhi(r7, r5); // 48
r0 = r0 ^ dataset[r2 & MASK]; // 49
r2 = r2 - r6; // 50
r7 = r7 - r5; // 51
r2 = r2 ^ r3; // 52
r7 = r7 - r0; // 53
r3 = r5 * r0 + r3; // 54
r7 = r7 ^ r5; // 55
r2 = r2 ^ dataset[r7 & MASK]; // 56
r5 = r5 - r6; // 57
r1 = r1 ^ dataset[r3 & MASK]; // 58
{ uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
r4 = r4 - r6; // 60
r1 = r1 * r2; // 61
r3 = r6 * r0 + r3; // 62
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
}
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
out[gid] = ((ulong)hi << 32) | (ulong)lo;
}
}

View file

@ -0,0 +1,57 @@
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset
#pragma once
#ifdef __cplusplus
#include <cstdint>
#else
#include <stdint.h>
#endif
#define IGNEUM_VEC_WARPS 3
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
{ // base nonce 0
0x82097370c10ee4eaull, 0xa0f16965a422e758ull, 0x2b7513e2593b170cull, 0x8fbb69f8974c4533ull, 0x500c384970d19f32ull, 0x72fb96e1820fa2b8ull, 0xca1521df6a934a65ull, 0x7407397e10548cbcull,
0xa619f127d1aa896bull, 0x36ce1104c1517d86ull, 0x6078ddd46d6d26b2ull, 0x2d22ded89010fbd3ull, 0xb030b743355bf6f9ull, 0x8280c8c1f4946ef7ull, 0xbfdae5712c7b9ad4ull, 0x50e9136a89e2e046ull,
0xfe0230d874a94f3aull, 0x8b6dbb8ff0546c32ull, 0x41438f3ab2d35cbcull, 0x6faf2dd5c5997f9cull, 0x52c3aae40019a0abull, 0xc455bf9926661472ull, 0x4eaf9e34af6c7c22ull, 0x9dcd05c3a3991115ull,
0x3a4f0ed406a66796ull, 0x7615703c08eabd8aull, 0x6007bcf34b8c6341ull, 0x1fb11900437441aeull, 0x540cff895991c6acull, 0x367a10121393f054ull, 0xf0dd7aa43fbdc2f4ull, 0x53d26f9c120f8a64ull
},
{ // base nonce 4096
0xc04adc4c89d1593bull, 0xc938ee393ee383dcull, 0xaef535444b6968fbull, 0x83581d97cebb4b30ull, 0x365315d0762bf586ull, 0xc26dfc2c40131920ull, 0x2f560c6fa351ead5ull, 0xc235996f6596d536ull,
0x1adf3384c9f23712ull, 0x0c87a8ad0dc872c3ull, 0x82471bfd2a3182d4ull, 0xe8828b0fb7560877ull, 0x3fd9044dd153dd7full, 0x3a91a9618ce2c525ull, 0xf8c503abac8481f9ull, 0x56b0989f6014f8dbull,
0xa1107a2732c588baull, 0xfc47a39e530d7efaull, 0x9237f7fc01727703ull, 0x01b5cbb25e6629c1ull, 0xae8e66fcec949f3eull, 0xb31655fbd75d3dbdull, 0x5a7750b585b3d0f1ull, 0x72f0484703cf30a0ull,
0x4cea6de3f76e970bull, 0x64fd914fb1fb5096ull, 0xbdcca26f07e7182cull, 0xc86ba79905164a5bull, 0x521b7d3ac36f06cbull, 0x6b8618a80cef2e71ull, 0x13df43bef1f7852full, 0x202e92ce7b597a33ull
},
{ // base nonce 1000000
0x4db5bf37f24811f2ull, 0x5e93450594a45e5full, 0xebe57bb921f163aaull, 0x0457e6f5ac702b1eull, 0x442d8926fceec0d4ull, 0x417256d19cce43d3ull, 0x3d61563a8303daceull, 0x4a43c5efac6f6d47ull,
0x9ef8a8d0a98f5c7eull, 0xbee88175926bb251ull, 0xb3b8e73f1e427be1ull, 0x515405b57446beecull, 0xba4f1765e616bf9aull, 0x148e5d9895c48299ull, 0x303fed05bdcdd7d1ull, 0x44d07bf31dba804full,
0x8616fd225f3851beull, 0x426a7a80ac1b462full, 0x6e0163361c5ec30full, 0x065b3666feb8d0e5ull, 0xf9bc697886c9983eull, 0x90bc61f358b511f4ull, 0xfa47afe197811c66ull, 0x12a39e4e67aa2e97ull,
0xde01f49ccab26a42ull, 0x2c6e837e74897413ull, 0x62d22c8acdeb1d09ull, 0x7fa8da035f65bb0bull, 0xdc4ce47ce0d48b6bull, 0x6199717653754041ull, 0x3a5e113c0d160d86ull, 0x718b3357f2391b60ull
}
};
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
static const uint32_t IGNEUM_DS_HEAD[16] = {
0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu,
0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du
};
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u;
// 64 sampled dataset words (index, value) computed on the Mac.
#define IGNEUM_DS_SAMPLES 64
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
};
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u
};
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u,
0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u
};
static const uint32_t IGNEUM_CACHE_LAST[16] = {
0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du,
0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu
};
static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull;

View file

@ -0,0 +1,36 @@
{
"seed": "igneum-genesis",
"day": "2026-10-03",
"dataset_mode": "memory-hard",
"dataset_log2_words": 28,
"mask": "0x0fffffff",
"lanes": 32,
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
"warps": [
{"base_nonce": 0, "expected": [
"0x82097370c10ee4ea", "0xa0f16965a422e758", "0x2b7513e2593b170c", "0x8fbb69f8974c4533", "0x500c384970d19f32", "0x72fb96e1820fa2b8", "0xca1521df6a934a65", "0x7407397e10548cbc",
"0xa619f127d1aa896b", "0x36ce1104c1517d86", "0x6078ddd46d6d26b2", "0x2d22ded89010fbd3", "0xb030b743355bf6f9", "0x8280c8c1f4946ef7", "0xbfdae5712c7b9ad4", "0x50e9136a89e2e046",
"0xfe0230d874a94f3a", "0x8b6dbb8ff0546c32", "0x41438f3ab2d35cbc", "0x6faf2dd5c5997f9c", "0x52c3aae40019a0ab", "0xc455bf9926661472", "0x4eaf9e34af6c7c22", "0x9dcd05c3a3991115",
"0x3a4f0ed406a66796", "0x7615703c08eabd8a", "0x6007bcf34b8c6341", "0x1fb11900437441ae", "0x540cff895991c6ac", "0x367a10121393f054", "0xf0dd7aa43fbdc2f4", "0x53d26f9c120f8a64"
]},
{"base_nonce": 4096, "expected": [
"0xc04adc4c89d1593b", "0xc938ee393ee383dc", "0xaef535444b6968fb", "0x83581d97cebb4b30", "0x365315d0762bf586", "0xc26dfc2c40131920", "0x2f560c6fa351ead5", "0xc235996f6596d536",
"0x1adf3384c9f23712", "0x0c87a8ad0dc872c3", "0x82471bfd2a3182d4", "0xe8828b0fb7560877", "0x3fd9044dd153dd7f", "0x3a91a9618ce2c525", "0xf8c503abac8481f9", "0x56b0989f6014f8db",
"0xa1107a2732c588ba", "0xfc47a39e530d7efa", "0x9237f7fc01727703", "0x01b5cbb25e6629c1", "0xae8e66fcec949f3e", "0xb31655fbd75d3dbd", "0x5a7750b585b3d0f1", "0x72f0484703cf30a0",
"0x4cea6de3f76e970b", "0x64fd914fb1fb5096", "0xbdcca26f07e7182c", "0xc86ba79905164a5b", "0x521b7d3ac36f06cb", "0x6b8618a80cef2e71", "0x13df43bef1f7852f", "0x202e92ce7b597a33"
]},
{"base_nonce": 1000000, "expected": [
"0x4db5bf37f24811f2", "0x5e93450594a45e5f", "0xebe57bb921f163aa", "0x0457e6f5ac702b1e", "0x442d8926fceec0d4", "0x417256d19cce43d3", "0x3d61563a8303dace", "0x4a43c5efac6f6d47",
"0x9ef8a8d0a98f5c7e", "0xbee88175926bb251", "0xb3b8e73f1e427be1", "0x515405b57446beec", "0xba4f1765e616bf9a", "0x148e5d9895c48299", "0x303fed05bdcdd7d1", "0x44d07bf31dba804f",
"0x8616fd225f3851be", "0x426a7a80ac1b462f", "0x6e0163361c5ec30f", "0x065b3666feb8d0e5", "0xf9bc697886c9983e", "0x90bc61f358b511f4", "0xfa47afe197811c66", "0x12a39e4e67aa2e97",
"0xde01f49ccab26a42", "0x2c6e837e74897413", "0x62d22c8acdeb1d09", "0x7fa8da035f65bb0b", "0xdc4ce47ce0d48b6b", "0x6199717653754041", "0x3a5e113c0d160d86", "0x718b3357f2391b60"
]}
],
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],
"dataset_last_index": 268435455,
"dataset_last": "0xa33ada72",
"dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}],
"cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"],
"cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"],
"cache_fnv1a64": "0x48c4f5bf24166b2e"
}

209
proto-metal/packbench.swift Normal file
View file

@ -0,0 +1,209 @@
// packbench: runs a program pack (igneum-pow export) on the Mac's Metal GPU from its files alone: memhard.metal (cache
// fill, dataset build), program.metal (igneum_hash), program.h (constants), vectors.json (the CPU reference's vectors).
// Read-width experiment, 5 October 2026 (docs/plans/read-width.md): the Swift bench generates its own programs and does
// not know the experiment's load classes; this harness runs whatever text the Rust emitter wrote, so Metal is checked
// against the Rust CPU reference and timed without a Swift mirror of the generator. One file, no packages.
//
// swiftc -O -target arm64-apple-macos11 -o packbench packbench.swift -framework Metal
// ./packbench --pack <dir> [--batches 5] [--batch-log2 24] [--group 256] [--warps 2048]
//
// Prints one RESULT line per run: vectors, cache and dataset checks, the batch fingerprint (FNV-1a 64 over the 2^B
// outputs at base nonce 0) and MH/s by wall and by GPU time. Variant 5 packs (IGNEUM_PERSISTENT_WARPS) are launched
// as --warps persistent warps with a 1 MiB scratch each; the batch is rounded to a multiple of 32 x warps.
import Foundation
import Metal
func nowMs() -> Double { return Double(DispatchTime.now().uptimeNanoseconds) / 1e6 }
func fail(_ m: String) -> Never { print("FAIL: \(m)"); exit(1) }
struct Opts { var pack = ""; var batches = 5; var batchLog2 = 24; var group = 256; var warps = 2048 }
var opts = Opts()
var args = Array(CommandLine.arguments.dropFirst())
while !args.isEmpty {
let a = args.removeFirst()
func next() -> String { if args.isEmpty { fail("missing value for \(a)") }; return args.removeFirst() }
switch a {
case "--pack": opts.pack = next()
case "--batches": opts.batches = Int(next())!
case "--batch-log2": opts.batchLog2 = Int(next())!
case "--group": opts.group = Int(next())!
case "--warps": opts.warps = Int(next())!
default: fail("unknown argument \(a)")
}
}
if opts.pack.isEmpty { fail("--pack <dir> is required") }
func readText(_ name: String) -> String {
guard let s = try? String(contentsOfFile: opts.pack + "/" + name, encoding: .utf8) else { fail("cannot read \(opts.pack)/\(name)") }
return s
}
let programH = readText("program.h")
func defineU32(_ name: String) -> UInt32? {
let pat = "#define \(name) ([0-9a-fA-Fx]+)"
guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
let v = String(programH[Range(m.range(at: 1), in: programH)!]).replacingOccurrences(of: "u", with: "")
if v.hasPrefix("0x") { return UInt32(v.dropFirst(2), radix: 16) }
return UInt32(v)
}
func defineStr(_ name: String) -> String? {
let pat = "#define \(name) \"([^\"]*)\""
guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
return String(programH[Range(m.range(at: 1), in: programH)!])
}
let datasetLog2 = Int(defineU32("IGNEUM_DATASET_LOG2") ?? 28)
let datasetMode = defineU32("IGNEUM_DATASET_MODE") ?? 1
if datasetMode != 1 { fail("packbench runs memory-hard packs only") }
let cacheLog2 = Int(defineU32("IGNEUM_CACHE_LOG2_WORDS") ?? 26)
let cacheSegments = Int(defineU32("IGNEUM_CACHE_SEGMENTS") ?? 65536)
let loadsPerHash = Int(defineU32("IGNEUM_LOADS_PER_HASH") ?? 128)
let bytesPerHash = Int(defineU32("IGNEUM_BYTES_PER_HASH") ?? UInt32(loadsPerHash * 4))
let scratchOps = Int(defineU32("IGNEUM_SCRATCH_OPS") ?? 0)
let persistent = (defineU32("IGNEUM_PERSISTENT_WARPS") ?? 0) == 1
let scratchWordsPerLane = Int(defineU32("IGNEUM_SCRATCH_WORDS_PER_LANE") ?? 8192)
let className = defineStr("IGNEUM_LOAD_CLASS") ?? "v2"
let seedString = defineStr("IGNEUM_SEED_STRING") ?? "?"
let programId = defineStr("IGNEUM_PROGRAM_ID") ?? ""
// vectors.json: bases, expected outputs, dataset head / last, cache fingerprint
let vj = try! JSONSerialization.jsonObject(with: Data(contentsOf: URL(fileURLWithPath: opts.pack + "/vectors.json"))) as! [String: Any]
func hex64(_ s: String) -> UInt64 { return UInt64(s.dropFirst(2), radix: 16)! }
func hex32(_ s: String) -> UInt32 { return UInt32(s.dropFirst(2), radix: 16)! }
let warpsJ = vj["warps"] as! [[String: Any]]
let vecBases = warpsJ.map { UInt32(($0["base_nonce"] as! NSNumber).uint64Value) }
let vecOuts = warpsJ.map { ($0["expected"] as! [String]).map(hex64) }
let dsHead = (vj["dataset_head"] as! [String]).map(hex32)
let dsLastIndex = UInt32((vj["dataset_last_index"] as! NSNumber).uint64Value)
let dsLast = hex32(vj["dataset_last"] as! String)
let cacheFnvWant = hex64(vj["cache_fnv1a64"] as! String)
guard let device = MTLCreateSystemDefaultDevice(), let queue = device.makeCommandQueue() else { fail("no Metal device") }
let words = 1 << datasetLog2
let mask = UInt32(words - 1)
let cacheWords = 1 << cacheLog2
func compile(_ file: String) -> MTLLibrary {
do { return try device.makeLibrary(source: readText(file), options: MTLCompileOptions()) } catch { fail("Metal compile of \(file): \(error)") }
}
let t0 = nowMs()
let mhLib = compile("memhard.metal")
let progLib = compile("program.metal")
guard let fillFn = mhLib.makeFunction(name: "igneum_cache_fill"), let buildFn = mhLib.makeFunction(name: "igneum_build"), let hashFn = progLib.makeFunction(name: "igneum_hash") else { fail("kernel functions missing") }
let fillPipe = try! device.makeComputePipelineState(function: fillFn)
let buildPipe = try! device.makeComputePipelineState(function: buildFn)
let hashPipe = try! device.makeComputePipelineState(function: hashFn)
let compileMs = nowMs() - t0
if hashPipe.threadExecutionWidth != 32 { print("WARNING: threadExecutionWidth \(hashPipe.threadExecutionWidth), not 32") }
guard let cache = device.makeBuffer(length: cacheWords * 4, options: .storageModePrivate) else { fail("cache alloc") }
guard let dataset = device.makeBuffer(length: words * 4, options: .storageModePrivate) else { fail("dataset alloc") }
func run(_ body: (MTLComputeCommandEncoder) -> Void) -> (Double, Double) {
let cb = queue.makeCommandBuffer()!
let enc = cb.makeComputeCommandEncoder()!
body(enc)
enc.endEncoding()
let w0 = nowMs()
cb.commit(); cb.waitUntilCompleted()
if let e = cb.error { fail("command buffer: \(e)") }
return (nowMs() - w0, (cb.gpuEndTime - cb.gpuStartTime) * 1000)
}
let (cacheWall, cacheGpu) = run { enc in
enc.setComputePipelineState(fillPipe); enc.setBuffer(cache, offset: 0, index: 0)
enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
}
let items = words / 16
let (buildWall, buildGpu) = run { enc in
enc.setComputePipelineState(buildPipe); enc.setBuffer(cache, offset: 0, index: 0); enc.setBuffer(dataset, offset: 0, index: 1)
enc.dispatchThreadgroups(MTLSize(width: items / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
}
// cache fingerprint and dataset head/last through a blit to shared memory
func blit(_ src: MTLBuffer, _ offset: Int, _ n: Int) -> MTLBuffer {
let dst = device.makeBuffer(length: n, options: .storageModeShared)!
let cb = queue.makeCommandBuffer()!; let b = cb.makeBlitCommandEncoder()!
b.copy(from: src, sourceOffset: offset, to: dst, destinationOffset: 0, size: n); b.endEncoding(); cb.commit(); cb.waitUntilCompleted()
return dst
}
func fnv1a64(_ p: UnsafeRawPointer, _ n: Int) -> UInt64 {
var h: UInt64 = 0xcbf29ce484222325
let b = p.bindMemory(to: UInt8.self, capacity: n)
for i in 0..<n { h ^= UInt64(b[i]); h = h &* 0x100000001b3 }
return h
}
let cacheCopy = blit(cache, 0, cacheWords * 4)
let cacheFnv = fnv1a64(cacheCopy.contents(), cacheWords * 4)
let cacheOk = cacheFnv == cacheFnvWant
let headCopy = blit(dataset, 0, 64)
let headPtr = headCopy.contents().bindMemory(to: UInt32.self, capacity: 16)
var dsOk = (0..<16).allSatisfy { headPtr[$0] == dsHead[$0] }
let lastCopy = blit(dataset, Int(dsLastIndex) * 4, 4)
dsOk = dsOk && lastCopy.contents().bindMemory(to: UInt32.self, capacity: 1)[0] == dsLast
// scratch arena (variant 5)
var scratch: MTLBuffer? = nil
var salt: UInt32 = 1
let warpsN = persistent ? opts.warps : 0
if persistent {
let bytes = warpsN * 32 * scratchWordsPerLane * 4
guard let s = device.makeBuffer(length: bytes, options: .storageModePrivate) else { fail("scratch alloc of \(bytes >> 20) MiB") }
scratch = s
}
// One hash launch: `nonces` outputs from `base`. Persistent: warpsN warps loop over nonces / 32 units.
func encodeHash(_ enc: MTLComputeCommandEncoder, out: MTLBuffer, base: UInt32, nonces: Int, group: Int) {
enc.setComputePipelineState(hashPipe)
enc.setBuffer(dataset, offset: 0, index: 0)
enc.setBuffer(out, offset: 0, index: 1)
var b = base; enc.setBytes(&b, length: 4, index: 2)
if persistent {
let units = nonces / 32
let nw = min(warpsN, units)
if units % nw != 0 { fail("nonces \(nonces) is not a multiple of 32 x \(nw) warps") }
enc.setBuffer(scratch!, offset: 0, index: 3)
var g = UInt32(units); enc.setBytes(&g, length: 4, index: 4)
var s = salt; enc.setBytes(&s, length: 4, index: 5)
salt = salt &+ UInt32(units)
let threads = nw * 32
let tg = min(group, threads)
enc.dispatchThreadgroups(MTLSize(width: threads / tg, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1))
} else {
let tg = min(group, nonces)
enc.dispatchThreadgroups(MTLSize(width: nonces / tg, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1))
}
}
// Vectors, standalone (one unit per launch)
var vecPass = 0
let vecOut = device.makeBuffer(length: 32 * 8, options: .storageModeShared)!
for (i, base) in vecBases.enumerated() {
_ = run { enc in encodeHash(enc, out: vecOut, base: base, nonces: 32, group: 32) }
let p = vecOut.contents().bindMemory(to: UInt64.self, capacity: 32)
var ok = true
for l in 0..<32 where p[l] != vecOuts[i][l] { ok = false; print("vector warp base \(base) lane \(l): GPU \(String(format: "%016llx", p[l])) expected \(String(format: "%016llx", vecOuts[i][l]))"); break }
if ok { vecPass += 1 }
}
// Batch at base 0: fingerprint and the vectors inside the batch
var nonces = 1 << opts.batchLog2
if persistent { let unit = 32 * min(warpsN, nonces / 32); nonces = (nonces / unit) * unit }
let out = device.makeBuffer(length: nonces * 8, options: .storageModeShared)!
let (warmWall, warmGpu) = run { enc in encodeHash(enc, out: out, base: 0, nonces: nonces, group: opts.group) }
let outPtr = out.contents().bindMemory(to: UInt64.self, capacity: nonces)
var batchVecPass = 0, batchVecN = 0
for (i, base) in vecBases.enumerated() where Int(base) + 32 <= nonces {
batchVecN += 1
if (0..<32).allSatisfy({ outPtr[Int(base) + $0] == vecOuts[i][$0] }) { batchVecPass += 1 }
}
let fingerprint = fnv1a64(out.contents(), nonces * 8)
// Timed batches
var wallSum = 0.0, gpuSum = 0.0
for b in 0..<opts.batches {
let (w, g) = run { enc in encodeHash(enc, out: out, base: UInt32(truncatingIfNeeded: (b + 1) * nonces), nonces: nonces, group: opts.group) }
wallSum += w; gpuSum += g
}
let hashes = Double(nonces) * Double(opts.batches)
let mhsWall = hashes / wallSum / 1e3, mhsGpu = hashes / gpuSum / 1e3
let packName = (opts.pack as NSString).lastPathComponent
print("pack \(packName) seed \"\(seedString)\" id \(programId) class \(className): loads/hash \(loadsPerHash), dataset bytes/hash \(bytesPerHash), scratch ops/hash \(scratchOps * 8)")
print("device \(device.name); compile \(String(format: "%.0f", compileMs)) ms; cache fill \(String(format: "%.1f", cacheGpu)) ms GPU (\(String(format: "%.1f", cacheWall)) wall); dataset build \(String(format: "%.1f", buildGpu)) ms GPU (\(String(format: "%.1f", buildWall)) wall)")
print("cache FNV-1a 64 \(String(format: "%016llx", cacheFnv)) \(cacheOk ? "PASS" : "FAIL"); dataset head and last \(dsOk ? "PASS" : "FAIL"); vectors standalone \(vecPass)/\(vecBases.count), in batch \(batchVecPass)/\(batchVecN)")
print("warm-up batch \(nonces) hashes: \(String(format: "%.1f", warmGpu)) ms GPU, \(String(format: "%.1f", warmWall)) ms wall")
let overall = cacheOk && dsOk && vecPass == vecBases.count && batchVecPass == batchVecN
print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) arena_mib=\(persistent ? warpsN : 0) nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")")
exit(overall ? 0 : 1)

View file

@ -41,7 +41,12 @@
#endif
// Kernels from kernel.cl (compiled as a separate C++ translation unit with the same defines).
#ifdef IGNEUM_PERSISTENT_WARPS
// Variant 5 of the read-width experiment: persistent warps with a 1 MiB scratch each (program.h says so).
void igneum_hash(const uint* ds, ulong* out, uint baseNonce, uint mask, uint* scratch, uint groups, uint salt);
#else
void igneum_hash(const uint* ds, ulong* out, uint baseNonce, uint mask);
#endif
#if IGNEUM_DATASET_MODE == 1
void igneum_cache_fill(uint* cache, uint nSegments);
void igneum_build(uint* ds, const uint* cache, uint nItems);
@ -280,14 +285,39 @@ int main(int argc, char** argv) {
// Vectors standalone: one work-group of IGNEUM_GROUP items per base nonce (the first 32 are the vector warp).
const uint32_t nonces = 1u << batchLog2;
std::vector<ulong> out(nonces);
#ifdef IGNEUM_PERSISTENT_WARPS
// Variant 5: EMU_WARPS persistent warps (one per work-group of 32), each with IGNEUM_SCRATCH_WORDS_PER_LANE x 32 words.
const unsigned emuWarps = 8;
std::vector<uint> scratchArena((size_t)emuWarps * 32u * IGNEUM_SCRATCH_WORDS_PER_LANE, 0u);
uint salt = 1u;
if (IGNEUM_GROUP != 32) { std::printf("variant 5 needs IGNEUM_GROUP 32 (one warp per work-group)\n"); return 2; }
std::printf("variant 5: %u persistent warps, scratch arena %u MiB, lazy tagged fill\n", emuWarps, (unsigned)((scratchArena.size() * 4u) >> 20));
auto launchHash = [&](size_t units, uint base) {
size_t nw = units < emuWarps ? units : emuWarps;
emu_launch(igneum_hash, nw * 32u, 32u, (const uint*)ds.data(), out.data(), base, mask, scratchArena.data(), (uint)units, salt);
salt += (uint)units;
};
#else
auto launchHash = [&](size_t nonceCount, uint base) {
emu_launch(igneum_hash, nonceCount, (unsigned)IGNEUM_GROUP, (const uint*)ds.data(), out.data(), base, mask);
};
#endif
for (int w = 0; w < IGNEUM_VEC_WARPS; ++w) {
emu_launch(igneum_hash, (size_t)IGNEUM_GROUP, (unsigned)IGNEUM_GROUP, (const uint*)ds.data(), out.data(), (uint)IGNEUM_VEC_BASE[w], mask);
#ifdef IGNEUM_PERSISTENT_WARPS
launchHash(1, (uint)IGNEUM_VEC_BASE[w]);
#else
launchHash((size_t)IGNEUM_GROUP, (uint)IGNEUM_VEC_BASE[w]);
#endif
char how[96];
std::snprintf(how, sizeof(how), "standalone, work-group %d, sub-group width %u", IGNEUM_GROUP, gSubGroupWidth);
overall = compareWarp((const uint64_t*)out.data(), IGNEUM_VEC_OUT[w], IGNEUM_VEC_BASE[w], how) && overall;
}
// In batch: every vector warp that fits in 2^batchLog2 nonces.
emu_launch(igneum_hash, (size_t)nonces, (unsigned)IGNEUM_GROUP, (const uint*)ds.data(), out.data(), 0u, mask);
#ifdef IGNEUM_PERSISTENT_WARPS
launchHash(nonces / 32u, 0u);
#else
launchHash((size_t)nonces, 0u);
#endif
for (int w = 0; w < IGNEUM_VEC_WARPS; ++w) {
if ((uint64_t)IGNEUM_VEC_BASE[w] + 32ull > nonces) { std::printf("verify warp base %u in batch: skipped (batch has %u nonces)\n", IGNEUM_VEC_BASE[w], nonces); continue; }
char how[96];

View file

@ -207,6 +207,9 @@ typedef struct {
int memprobe; // --memprobe: dependent-load latency and throughput, independent-load throughput and an ALU
// chain on the chosen device, no pack needed (5 October 2026, the 9070 XT on the eGPU)
int probeMib; // --probe-mib N: --memprobe at that one buffer size only (default 0 = 4, 64 and 1024 MiB)
int benchPack; // --bench-pack: with --pack D, build and self-test the pack at run time (as --serve does) and time
// igneum_hash_bound with the pack's seed words as init words; read-width experiment, 5 October 2026
int warps; // --warps N: persistent warps for a variant-5 pack (IGNEUM_PERSISTENT_WARPS); 0 = 2048
} Options;
static int packMib(void) { return (int)(((1ull << IGNEUM_DATASET_LOG2) * 4ull) >> 20); }
@ -236,6 +239,8 @@ static void usage(void) {
" pack this exe was built against; it is self-tested against its vectors.h first (the one-click worker)\n"
" --readback M with --serve: select (default) reads back only the hits and 34 sentinel words of each dispatch through a\n"
" GPU-side pass; full reads back every output (8 bytes per nonce). IGNEUM_READBACK=full does the same.\n"
" --bench-pack with --pack D: read the pack at run time, build and self-test it, time its bound kernel (one exe, any pack)\n"
" --warps N persistent warps for a variant-5 pack (a 1 MiB scratch each; default 2048; the batch rounds to 32 x N)\n"
" --memprobe no pack: dependent random loads (latency and throughput against lanes in flight), independent random\n"
" loads and an ALU chain on the chosen device, at 4, 64 and 1024 MiB (--probe-mib N for one size)\n", packMib(), IGNEUM_KERNEL_PATH);
}
@ -248,7 +253,7 @@ static Options parseArgs(int argc, char** argv) {
int i;
o.datasetMib = 1024; o.batchLog2 = 24; o.batches = 5; o.groupWarps = 1; o.sweep = 0; o.device = -1;
o.exchange = 0; o.list = 0; o.timeWall = -1; o.kernelPath = IGNEUM_KERNEL_PATH; o.extraOpts = ""; o.serve = 0; o.noPrepare = 0; o.kernelGiven = 0; o.vendor = NULL; o.packDir = NULL;
o.readback = (getenv("IGNEUM_READBACK") && strcmp(getenv("IGNEUM_READBACK"), "full") == 0) ? 1 : 0; o.memprobe = 0; o.probeMib = 0;
o.readback = (getenv("IGNEUM_READBACK") && strcmp(getenv("IGNEUM_READBACK"), "full") == 0) ? 1 : 0; o.memprobe = 0; o.probeMib = 0; o.benchPack = 0; o.warps = 0;
for (i = 1; i < argc; ++i) {
const char* a = argv[i];
int needs = (strcmp(a, "--dataset-mib") == 0 || strcmp(a, "--batch-log2") == 0 || strcmp(a, "--batches") == 0 ||
@ -274,6 +279,8 @@ static Options parseArgs(int argc, char** argv) {
else { printf("--readback must be select or full\n"); exit(2); }
}
else if (strcmp(a, "--memprobe") == 0) o.memprobe = 1;
else if (strcmp(a, "--bench-pack") == 0) o.benchPack = 1;
else if (strcmp(a, "--warps") == 0) { if (i + 1 >= argc) { usage(); exit(2); } o.warps = atoi(argv[++i]); }
else if (strcmp(a, "--probe-mib") == 0) { if (i + 1 >= argc) { usage(); exit(2); } o.probeMib = atoi(argv[++i]); }
else if (strcmp(a, "--build-opts") == 0) o.extraOpts = argv[++i];
else if (strcmp(a, "--time") == 0) {
@ -1013,6 +1020,19 @@ static int unhexBuf(const char* s, uint8_t* out, size_t cap, size_t* len) {
* from the compiled-in program.h, so one prebuilt exe serves every pack. The compiled-in values are the defaults. */
static PfPack gPack;
static int gGeneric = 0;
/* Variant 5 of the read-width experiment (5 October 2026): the scratch arena of a persistent-warp pack, its warp count
* and the running tag salt; set by --bench-pack before the self-test. Serve mode does not support these packs. */
static cl_mem gScratch = NULL;
static cl_uint gScratchWarps = 0;
static cl_uint gSalt = 1;
/* Sets the three extra arguments of a variant-5 kernel (after the five of igneum_hash_bound) for `units` units. */
static cl_int setScratchArgs(cl_kernel k, cl_uint firstArg, cl_uint units) {
cl_int e = clSetKernelArg(k, firstArg, sizeof(cl_mem), &gScratch);
if (e == CL_SUCCESS) e = clSetKernelArg(k, firstArg + 1, sizeof(cl_uint), &units);
if (e == CL_SUCCESS) e = clSetKernelArg(k, firstArg + 2, sizeof(cl_uint), &gSalt);
gSalt += units;
return e;
}
static uint32_t gServeWords = 1u << IGNEUM_DATASET_LOG2;
#if IGNEUM_DATASET_MODE == 1
static uint32_t gServeCacheWords = 1u << IGNEUM_CACHE_LOG2_WORDS;
@ -1102,6 +1122,10 @@ static int pairSelfTest(Device* dv, const DeviceInfo* di, cl_command_queue q, Se
out = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, g * sizeof(uint64_t), NULL, &e);
if (e == CL_SUCCESS) { ++gMemCreated; init = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY | CL_MEM_COPY_HOST_PTR, 32, pk.seedw, &e); }
if (e == CL_SUCCESS) ++gMemCreated;
if (pk.persistent) {
if (!gScratch) { snprintf(err, errCap, "self-test: a variant-5 pack (persistent warps) needs --bench-pack (serve mode does not carry a scratch)"); return 0; }
g = 32; local = 32; /* one persistent warp runs the one unit */
}
for (w = 0; w < pk.vecWarps && e == CL_SUCCESS; ++w) {
cl_uint base = pk.vecBase[w], mask = words - 1u;
e = clSetKernelArg(p->kHashBound, 0, sizeof(cl_mem), &p->ds);
@ -1109,6 +1133,7 @@ static int pairSelfTest(Device* dv, const DeviceInfo* di, cl_command_queue q, Se
if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 2, sizeof(cl_uint), &base);
if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 3, sizeof(cl_uint), &mask);
if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 4, sizeof(cl_mem), &init);
if (e == CL_SUCCESS && pk.persistent) e = setScratchArgs(p->kHashBound, 5, 1u);
if (e == CL_SUCCESS) e = clEnqueueNDRangeKernel(q, p->kHashBound, 1, NULL, &g, &local, 0, NULL, NULL);
if (e == CL_SUCCESS) e = clEnqueueReadBuffer(q, out, CL_TRUE, 0, 32 * sizeof(uint64_t), &vec[w * 32], 0, NULL, NULL);
}
@ -1209,6 +1234,92 @@ static int startPrepareThread(PrepareTask* t) { pthread_t th; if (pthread_create
#endif
#endif
/* --bench-pack (read-width experiment, 5 October 2026): the pack in --pack is built and self-tested exactly as the
* first pair of --serve (pairBuffers: cache, dataset, cache FNV, dataset words, the vector warps through
* igneum_hash_bound with the pack's seed words), then the bound kernel is timed over --batches dispatches of
* 2^--batch-log2 nonces with device event time, and the 2^B outputs at base nonce 0 are fingerprinted (FNV-1a 64) so
* the same pack can be compared bit for bit across vendors. A variant-5 pack is launched as --warps persistent warps
* with a 1 MiB scratch each. One line per run starts with RESULT. */
static int runBenchPack(Device* dv, const DeviceInfo* di, const Options* o) {
#if IGNEUM_DATASET_MODE != 1
(void)dv; (void)di; (void)o;
printf("FAIL: --bench-pack needs a memory-hard placeholder pack\n");
return 2;
#else
const uint32_t words = gServeWords, mask = words - 1u;
uint32_t nonces = 1u << o->batchLog2;
size_t groupSize = 32 * (size_t)o->groupWarps, g;
cl_int err = 0;
cl_mem dOut, dInit;
uint64_t* hOut;
ServePair* cur;
char perr[512], devName[256];
double t0 = wallMs(), sum = 0, warmMs;
uint64_t fp;
int b, k;
cl_uint warps = (cl_uint)(o->warps > 0 ? o->warps : 2048), units = nonces / 32u;
if (!dv->kHashBound) { printf("FAIL: the kernel source has no igneum_hash_bound\n"); return 2; }
if (gPack.persistent) {
size_t arena;
if (o->groupWarps != 1) { printf("FAIL: a variant-5 pack needs --group-warps 1 (one warp per work-group: the loop trip count must be uniform)\n"); return 2; }
while (warps > 1 && units % warps != 0) warps >>= 1;
arena = (size_t)warps * 32u * (size_t)gPack.scratchWordsPerLane * 4u;
if ((uint64_t)arena > di->maxAlloc) { printf("FAIL: scratch arena %llu MiB exceeds the device's max alloc %llu MiB; lower --warps\n", (unsigned long long)(arena >> 20), (unsigned long long)(di->maxAlloc >> 20)); return 2; }
gScratch = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, arena, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer scratch");
gScratchWarps = warps;
printf("variant 5: %u persistent warps (%u x 32 work-items, work-group 32), scratch arena %llu MiB, %u units per dispatch, lazy tagged fill\n", warps, warps, (unsigned long long)(arena >> 20), units);
}
cur = (ServePair*)calloc(1, sizeof(ServePair));
cur->kHashBound = dv->kHashBound; cur->kCacheFill = dv->kCacheFill; cur->kBuild = dv->kBuild; cur->prog = dv->prog;
dv->kHashBound = dv->kCacheFill = dv->kBuild = NULL; dv->prog = NULL;
memcpy(cur->sw, gPack.seedw, 32); memcpy(cur->kw, gPack.keyw, 32);
if (!pairBuffers(dv, di, dv->q, cur, words, gServeCacheWords, gServeSegments, o->packDir, perr, sizeof(perr))) { printf("FAIL: pack %s: %s\n", o->packDir, perr); return 1; }
printf("pack %s: cache %.0f dataset %.0f check %.0f ms (%.0f ms in all); %s\n", o->packDir, cur->cacheMs, cur->datasetMs, cur->checkMs, wallMs() - t0, cur->check);
printKernelInfo(di, cur->kHashBound, "igneum_hash_bound", (int)groupSize, "");
dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)nonces * sizeof(uint64_t), NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer out");
dInit = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY | CL_MEM_COPY_HOST_PTR, 32, cur->sw, &err); CL_CHECK_ERR(err, "clCreateBuffer init words");
hOut = (uint64_t*)malloc((size_t)nonces * sizeof(uint64_t));
strncpy(devName, di->name, 255); devName[255] = 0;
for (k = 0; devName[k]; ++k) if (devName[k] == ' ') devName[k] = '_';
g = gPack.persistent ? (size_t)warps * 32u : (size_t)nonces;
for (b = -1; b < o->batches; ++b) {
cl_uint base = (cl_uint)((uint32_t)(b + 1) * nonces);
cl_event ev = NULL;
double ms;
CL_CHECK(clSetKernelArg(cur->kHashBound, 0, sizeof(cl_mem), &cur->ds));
CL_CHECK(clSetKernelArg(cur->kHashBound, 1, sizeof(cl_mem), &dOut));
CL_CHECK(clSetKernelArg(cur->kHashBound, 2, sizeof(cl_uint), &base));
CL_CHECK(clSetKernelArg(cur->kHashBound, 3, sizeof(cl_uint), &mask));
CL_CHECK(clSetKernelArg(cur->kHashBound, 4, sizeof(cl_mem), &dInit));
if (gPack.persistent) CL_CHECK(setScratchArgs(cur->kHashBound, 5, units));
{
double w0 = wallMs();
CL_CHECK(clEnqueueNDRangeKernel(dv->q, cur->kHashBound, 1, NULL, &g, &groupSize, 0, NULL, &ev));
CL_CHECK(clWaitForEvents(1, &ev));
ms = o->timeWall ? wallMs() - w0 : eventMs(ev);
if (ms < 0) ms = wallMs() - w0;
clReleaseEvent(ev);
}
if (b < 0) {
warmMs = ms;
CL_CHECK(clEnqueueReadBuffer(dv->q, dOut, CL_TRUE, 0, (size_t)nonces * sizeof(uint64_t), hOut, 0, NULL, NULL));
fp = pf_fnv1a64((const uint32_t*)hOut, (size_t)nonces * 8u);
} else sum += ms;
}
printf("warm-up dispatch (base 0): %.2f ms; %d timed dispatches of 2^%d nonces: mean %.2f ms\n", warmMs, o->batches, o->batchLog2, sum / o->batches);
printf("RESULT pack=%s class=%s device=%s platform=%s group=%d warps=%u arena_mib=%llu nonces=%u batches=%d check=%s fingerprint=%016llx mhs=%.3f loads=%u bytes=%u scratch_ops=%u time=%s\n",
o->packDir, gPack.loadClass, devName, strcmp(di->platformName, "Apple") == 0 ? "Apple" : "other", (int)groupSize, gPack.persistent ? warps : 0u,
gPack.persistent ? (unsigned long long)(((size_t)warps * 32u * gPack.scratchWordsPerLane * 4u) >> 20) : 0ull, nonces, o->batches,
cur->checked ? "PASS" : "skipped", (unsigned long long)fp, (double)nonces * (double)o->batches / (sum / 1000.0) / 1e6,
gPack.loadsPerHash, gPack.bytesPerHash, gPack.scratchOps * 8u, o->timeWall ? "wall" : "event");
free(hOut);
clReleaseMemObject(dOut); clReleaseMemObject(dInit);
if (gScratch) clReleaseMemObject(gScratch);
releasePair(cur);
return 0;
#endif
}
static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
#if IGNEUM_DATASET_MODE != 1
(void)dv; (void)di; (void)o;
@ -1612,6 +1723,11 @@ static const char* PROBE_SRC =
" }\n"
" out[g] = x0 ^ x1 ^ x2 ^ x3 ^ x4 ^ x5 ^ x6 ^ x7;\n"
"}\n"
"__kernel void probe_line16(__global const uint4* ds, uint vecMask, uint steps, uint seed, __global uint* out) {\n"
" uint x = pm_mix((uint)get_global_id(0) ^ seed);\n"
" for (uint s = 0u; s < steps; ++s) { uint4 a = ds[x & vecMask]; x = (a.x ^ a.y ^ a.z ^ a.w) ^ (x * 0x9E3779B1u + s); }\n"
" out[get_global_id(0)] = x;\n"
"}\n"
"__kernel void probe_line(__global const uint4* ds, uint lineMask, uint steps, uint seed, __global uint* out) {\n"
" uint x = pm_mix((uint)get_global_id(0) ^ seed);\n"
" for (uint s = 0u; s < steps; ++s) {\n"
@ -1656,7 +1772,7 @@ static double probeLaunch(Device* dv, const Options* o, cl_kernel k, size_t glob
static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) {
cl_int err = 0;
cl_program prog;
cl_kernel kFill, kChase, kIndep, kAlu, kLine, kStream;
cl_kernel kFill, kChase, kIndep, kAlu, kLine, kStream, kLine16;
size_t srcLen = strlen(PROBE_SRC);
int sizes[3] = { 4, 64, 1024 }, nSizes = 3, si;
size_t lanesList[8] = { 256, 1024, 1u << 12, 1u << 14, 1u << 16, 1u << 18, 1u << 20, 1u << 22 };
@ -1682,6 +1798,7 @@ static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) {
kIndep = clCreateKernel(prog, "probe_indep", &err); CL_CHECK_ERR(err, "probe_indep");
kAlu = clCreateKernel(prog, "probe_alu", &err); CL_CHECK_ERR(err, "probe_alu");
kLine = clCreateKernel(prog, "probe_line", &err); CL_CHECK_ERR(err, "probe_line");
kLine16 = clCreateKernel(prog, "probe_line16", &err); CL_CHECK_ERR(err, "probe_line16");
kStream = clCreateKernel(prog, "probe_stream", &err); CL_CHECK_ERR(err, "probe_stream");
dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, maxLanes * 4u, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer probe out");
printf("memprobe on [%s] %s, driver %s, %u compute units, %u MHz, %s time\n", di->platformName, di->name, di->driver, di->computeUnits, di->clockMHz, o->timeWall ? "wall" : "device event");
@ -1737,6 +1854,25 @@ static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) {
fflush(stdout);
}
}
{
/* Random 16-byte reads (one uint4) in a dependent chain: the W = 16 width of the read-width experiment. */
size_t local = di->maxWorkGroup < 256 ? di->maxWorkGroup : 256;
size_t lanes;
cl_uint vecMask = (words / 4u) - 1u;
for (lanes = 1u << 14; lanes <= maxLanes; lanes <<= 2) {
cl_uint seed = 0x2718281u;
double ms;
CL_CHECK(clSetKernelArg(kLine16, 0, sizeof(cl_mem), &dDs));
CL_CHECK(clSetKernelArg(kLine16, 1, sizeof(cl_uint), &vecMask));
CL_CHECK(clSetKernelArg(kLine16, 2, sizeof(cl_uint), &STEPS));
CL_CHECK(clSetKernelArg(kLine16, 3, sizeof(cl_uint), &seed));
CL_CHECK(clSetKernelArg(kLine16, 4, sizeof(cl_mem), &dOut));
ms = probeLaunch(dv, o, kLine16, lanes, local, 3, 3, seed);
printf("| line 16 B | %d | %llu | %llu | %u | %.3f | %.3f G reads/s | %.1f GB/s in 16 B reads |\n", mib, (unsigned long long)local, (unsigned long long)lanes, STEPS, ms,
(double)lanes * (double)STEPS / (ms / 1000.0) / 1e9, (double)lanes * (double)STEPS * 16.0 / (ms / 1000.0) / 1e9);
fflush(stdout);
}
}
{
/* Random 64-byte lines (16 words, four uint4 loads) in a dependent chain: lines per second against the
* 4-byte chase above says what one random 4-byte read costs the memory system. If the two rates are
@ -1792,7 +1928,7 @@ static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) {
(double)lanes * (double)ALU_STEPS / (ms / 1000.0) / 1e9 / (double)(di->computeUnits ? di->computeUnits : 1));
}
clReleaseMemObject(dOut);
clReleaseKernel(kFill); clReleaseKernel(kChase); clReleaseKernel(kIndep); clReleaseKernel(kAlu); clReleaseKernel(kLine); clReleaseKernel(kStream);
clReleaseKernel(kFill); clReleaseKernel(kChase); clReleaseKernel(kIndep); clReleaseKernel(kAlu); clReleaseKernel(kLine); clReleaseKernel(kStream); clReleaseKernel(kLine16);
clReleaseProgram(prog);
printf("memprobe: done\n");
return 0;
@ -1821,7 +1957,7 @@ int main(int argc, char** argv) {
static char boundPath[1200];
char perr[512];
size_t n = strlen(o.packDir);
if (!o.serve) { printf("FAIL: --pack goes with --serve (the bench runs the compiled-in pack)\n"); return 2; }
if (!o.serve && !o.benchPack) { printf("FAIL: --pack goes with --serve or --bench-pack (the plain bench runs the compiled-in pack)\n"); return 2; }
if (n > 1 && (o.packDir[n - 1] == '/' || o.packDir[n - 1] == '\\')) ((char*)o.packDir)[n - 1] = 0;
if (!pf_load(o.packDir, &gPack, perr, sizeof(perr))) { printf("error 0 pack %s: %s\n", o.packDir, perr); fflush(stdout); return 2; }
gGeneric = 1;
@ -1877,6 +2013,7 @@ int main(int argc, char** argv) {
clReleaseContext(dv.ctx);
return rc;
}
if (o.benchPack && !o.packDir) { printf("FAIL: --bench-pack needs --pack <dir>\n"); return 2; }
if (o.serve && !o.kernelGiven) {
/* The bound kernel lives next to the compiled-in kernel.cl as kernel_bound.cl (packs from igneum-pow or igneum-miner export-pack). */
static char boundPath[1024];
@ -1894,6 +2031,7 @@ int main(int argc, char** argv) {
printf("build options: %s\n", dv.buildOptions);
printf("exchange: %s\n", dv.exchangeNote);
if (o.serve) return runServe(&dv, di, &o);
if (o.benchPack) return runBenchPack(&dv, di, &o);
printKernelInfo(di, dv.kHash, "igneum_hash", dv.groupSize, "");
printf("program: %d instructions x %d iterations, loads/hash %d, op mix %s\n",
IGNEUM_INSTR_COUNT, IGNEUM_ITERATIONS, IGNEUM_LOADS_PER_HASH, IGNEUM_OP_MIX);