read-width: scratch per warp is a class parameter (32 or 128 KiB, under the 6 GB working-set cap), distinct-address rule bounds dataset loads only; Metal pack harness; OpenCL --bench-pack, scratch args and 16-byte probe; packfile class fields; OpenCL emulator persistent launch
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
36d93e03c8
commit
ea863d6d4b
92 changed files with 6031 additions and 367 deletions
|
|
@ -14,7 +14,7 @@
|
|||
//! costs about a millisecond on one core. The census (section 7.3) checked on 100,000 programs that the
|
||||
//! closed-form verdict agrees with the memory-hard one on all but 39 threshold-edge cases.
|
||||
|
||||
use crate::generator::{Instr, Op, Program, INSTR_COUNT, ITERATIONS, LANES, SCRATCH_SLOT_MASK};
|
||||
use crate::generator::{Instr, Op, Program, INSTR_COUNT, ITERATIONS, LANES};
|
||||
use crate::seed::{fnv1a64, SplitMix64};
|
||||
use crate::verify::{dataset_elem, fold_words, splitmix32, ScratchModel};
|
||||
|
||||
|
|
@ -33,8 +33,10 @@ pub const BIAS_TOLERANCE: u32 = 136;
|
|||
/// Distinct addresses per lane per evaluation, summed over 2,048 evaluations, must exceed this (mean above 120).
|
||||
pub const MIN_DISTINCT_SUM: u64 = 245_760;
|
||||
|
||||
/// The distinct-address bound for a program with `loads` loads per hash: the same 120 of 128 ratio, so
|
||||
/// The distinct-address bound for a program with `loads` dataset loads per hash: the same 120 of 128 ratio, so
|
||||
/// [`MIN_DISTINCT_SUM`] for the lottery hash and `loads x 1,920` for the read-width classes with other counts.
|
||||
/// Variant 5's scratch read-modify-writes are not dataset loads: their slots repeat by design (a later
|
||||
/// read-modify-write sees an earlier write), so they are neither counted nor bounded here.
|
||||
pub fn min_distinct_sum(loads: usize) -> u64 {
|
||||
loads as u64 * ACCEPT_HASHES as u64 * 120 / 128
|
||||
}
|
||||
|
|
@ -73,7 +75,7 @@ impl std::fmt::Display for Reject {
|
|||
Reject::Saturated { count } => write!(f, "(c) {count} of 16384 final register values saturated (limit 163)"),
|
||||
Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"),
|
||||
Reject::DistinctAddresses { sum } => {
|
||||
write!(f, "(c) distinct addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 loads)", *sum as f64 / 2048.0)
|
||||
write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -189,7 +191,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
|
|||
}
|
||||
let mut idx = [0u32; LANES];
|
||||
let mut nload = 0usize;
|
||||
let mut scratch = if p.has_scratch() { Some(ScratchModel::new()) } else { None };
|
||||
let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None };
|
||||
let slot_mask = p.class.scratch_slot_mask();
|
||||
for it in 0..ITERATIONS {
|
||||
let sel = r[0];
|
||||
for (k, ins) in p.instrs.iter().enumerate() {
|
||||
|
|
@ -200,7 +203,7 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
|
|||
// Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word).
|
||||
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
|
||||
for lane in 0..LANES {
|
||||
idx[lane] = r[a][lane] & SCRATCH_SLOT_MASK;
|
||||
idx[lane] = r[a][lane] & slot_mask;
|
||||
}
|
||||
if idx.iter().all(|&x| x == idx[0]) {
|
||||
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
|
||||
|
|
@ -332,7 +335,8 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
|
|||
sl.sort_unstable();
|
||||
let mut distinct = 0u64;
|
||||
for k in 0..loads {
|
||||
if k == 0 || sl[k] != sl[k - 1] {
|
||||
// scratch slots carry bit 31 (variant 5) and are not dataset addresses
|
||||
if sl[k] & 0x8000_0000 == 0 && (k == 0 || sl[k] != sl[k - 1]) {
|
||||
distinct += 1;
|
||||
}
|
||||
}
|
||||
|
|
@ -367,7 +371,7 @@ pub fn check_dynamic(p: &Program) -> Result<AcceptReport, Reject> {
|
|||
}
|
||||
bias_max = bias_max.max(d);
|
||||
}
|
||||
if acc.distinct_sum <= min_distinct_sum(loads) {
|
||||
if acc.distinct_sum <= min_distinct_sum(loads - p.scratch_ops_per_hash()) {
|
||||
return Err(Reject::DistinctAddresses { sum: acc.distinct_sum });
|
||||
}
|
||||
Ok(AcceptReport { distinct_sum: acc.distinct_sum, saturated: acc.saturated, bias_max })
|
||||
|
|
@ -395,7 +399,7 @@ mod tests {
|
|||
/// with `verify.rs` on every class (the fold is shared, the addresses are aligned the same way).
|
||||
#[test]
|
||||
fn classes_pass_and_match_verify() {
|
||||
for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2", "scr8"] {
|
||||
for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2k32", "scr8k128"] {
|
||||
let c = LoadClass::parse(name).unwrap();
|
||||
let p = generate_class("igneum-genesis", c);
|
||||
assert!(check(&p).is_ok(), "{name}");
|
||||
|
|
|
|||
|
|
@ -15,7 +15,6 @@ use crate::memhard::{
|
|||
CACHE_TAG, CACHE_WORDS, CHACHA_ROUNDS, CHACHA_SIGMA, ITEM_ROUNDS,
|
||||
};
|
||||
use crate::seed::SplitMix64;
|
||||
use crate::generator::{SCRATCH_BYTES_PER_WARP, SCRATCH_SLOTS, SCRATCH_SLOT_MASK, SCRATCH_WORDS_PER_LANE};
|
||||
use crate::verify::{DatasetMode, DatasetSource, Epoch, FOLD_MUL, FOLD_ROT};
|
||||
|
||||
/// Where the words of a wide load come from (read-width experiment).
|
||||
|
|
@ -127,15 +126,11 @@ fn scratch_prelude(p: &Program, dialect: CoreDialect) -> String {
|
|||
CoreDialect::OpenCl => ("uint", "static inline"),
|
||||
};
|
||||
let mut s = String::new();
|
||||
s.push_str(&format!("// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, {} slots of
|
||||
", SCRATCH_SLOTS));
|
||||
s.push_str("// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
");
|
||||
s.push_str("// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
");
|
||||
s.push_str(&format!("// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a {} KiB scratch per warp, {} slots of\n", p.class.scratch_kb, p.class.scratch_slots_per_lane()));
|
||||
s.push_str("// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not\n");
|
||||
s.push_str("// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.\n");
|
||||
s.push_str(&format!(
|
||||
"{fn_} {u} scr_fill({u} gbase, {u} lane, {u} slot, {u} j) {{ {u} sw = (j == 0u) ? {} : ((j == 1u) ? {} : {}); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }}
|
||||
",
|
||||
"{fn_} {u} scr_fill({u} gbase, {u} lane, {u} slot, {u} j) {{ {u} sw = (j == 0u) ? {} : ((j == 1u) ? {} : {}); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }}\n",
|
||||
hex(p.seed[0]),
|
||||
hex(p.seed[1]),
|
||||
hex(p.seed[2])
|
||||
|
|
@ -145,15 +140,14 @@ fn scratch_prelude(p: &Program, dialect: CoreDialect) -> String {
|
|||
|
||||
/// Variant 5: one scratch read-modify-write as a statement block. `arena`, `tag`, `gbase` and `lane` are in scope
|
||||
/// (the persistent prologue). Reads 16 bytes, folds the three data words into dst, rewrites the slot behind the tag.
|
||||
fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str) -> String {
|
||||
fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str, slot_mask: u32) -> String {
|
||||
let (u, load, store) = match dialect {
|
||||
CoreDialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"),
|
||||
CoreDialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"),
|
||||
CoreDialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena);"),
|
||||
};
|
||||
format!(
|
||||
"{{ {u} s_ = {a} & {}u; {load} {u} m_ = (v_.x == tag) ? 0xffffffffu : 0u; {u} w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); {u} w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); {u} w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); {u} x_ = {d} ^ w0_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w1_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w2_; {d} = x_; {store} }}",
|
||||
SCRATCH_SLOT_MASK,
|
||||
"{{ {u} s_ = {a} & {slot_mask}u; {load} {u} m_ = (v_.x == tag) ? 0xffffffffu : 0u; {u} w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); {u} w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); {u} w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); {u} x_ = {d} ^ w0_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w1_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w2_; {d} = x_; {store} }}",
|
||||
k = hex(FOLD_MUL)
|
||||
)
|
||||
}
|
||||
|
|
@ -162,29 +156,21 @@ fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str) -> String {
|
|||
/// choice); warp `w` owns arena `w` and runs the units `w, w + N, w + 2N, ...` of the launch. Inside the loop the
|
||||
/// lottery hash's text is unchanged: `gid` is the unit's first output index plus the lane. The host MUST launch
|
||||
/// `groups` as a multiple of N (a uniform trip count: the OpenCL local-memory exchange carries a barrier).
|
||||
fn persistent_prologue(dialect: CoreDialect) -> String {
|
||||
fn persistent_prologue(dialect: CoreDialect, words_per_lane: usize) -> String {
|
||||
let (u, tid, nthreads, ptr) = match dialect {
|
||||
CoreDialect::Metal => ("uint", "tid", "nthreads", "device uint*"),
|
||||
CoreDialect::Cuda => ("uint32_t", "(blockIdx.x * blockDim.x + threadIdx.x)", "(gridDim.x * blockDim.x)", "uint32_t*"),
|
||||
CoreDialect::OpenCl => ("uint", "(uint)get_global_id(0)", "(uint)get_global_size(0)", "__global uint*"),
|
||||
};
|
||||
let mut s = String::new();
|
||||
s.push_str(&format!(" {u} lane = {tid} & 31u;
|
||||
"));
|
||||
s.push_str(&format!(" {u} warp_ = {tid} >> 5;
|
||||
"));
|
||||
s.push_str(&format!(" {u} nwarps_ = {nthreads} >> 5;
|
||||
"));
|
||||
s.push_str(&format!(" {ptr} arena = scratch + ((size_t)warp_ * 32u + lane) * {}u;
|
||||
", SCRATCH_WORDS_PER_LANE));
|
||||
s.push_str(&format!(" for ({u} g_ = warp_; g_ < groups; g_ += nwarps_) {{
|
||||
"));
|
||||
s.push_str(&format!(" {u} gid = g_ * 32u + lane;
|
||||
"));
|
||||
s.push_str(&format!(" {u} gbase = baseNonce + g_ * 32u;
|
||||
"));
|
||||
s.push_str(&format!(" {u} tag = salt + g_;
|
||||
"));
|
||||
s.push_str(&format!(" {u} lane = {tid} & 31u;\n"));
|
||||
s.push_str(&format!(" {u} warp_ = {tid} >> 5;\n"));
|
||||
s.push_str(&format!(" {u} nwarps_ = {nthreads} >> 5;\n"));
|
||||
s.push_str(&format!(" {ptr} arena = scratch + ((size_t)warp_ * 32u + lane) * {words_per_lane}u;\n"));
|
||||
s.push_str(&format!(" for ({u} g_ = warp_; g_ < groups; g_ += nwarps_) {{\n"));
|
||||
s.push_str(&format!(" {u} gid = g_ * 32u + lane;\n"));
|
||||
s.push_str(&format!(" {u} gbase = baseNonce + g_ * 32u;\n"));
|
||||
s.push_str(&format!(" {u} tag = salt + g_;\n"));
|
||||
s
|
||||
}
|
||||
|
||||
|
|
@ -194,20 +180,13 @@ fn scratch_header_lines(p: &Program) -> String {
|
|||
return String::new();
|
||||
}
|
||||
let mut s = String::new();
|
||||
s.push_str("// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
");
|
||||
s.push_str("// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
|
||||
");
|
||||
s.push_str("#define IGNEUM_PERSISTENT_WARPS 1
|
||||
");
|
||||
s.push_str(&format!("#define IGNEUM_SCRATCH_OPS {} // scratch read-modify-writes per program ({} per hash)
|
||||
", p.class.scratch_slots(), p.scratch_ops_per_hash()));
|
||||
s.push_str(&format!("#define IGNEUM_SCRATCH_SLOTS {SCRATCH_SLOTS}u
|
||||
"));
|
||||
s.push_str(&format!("#define IGNEUM_SCRATCH_WORDS_PER_LANE {SCRATCH_WORDS_PER_LANE}u
|
||||
"));
|
||||
s.push_str(&format!("#define IGNEUM_SCRATCH_BYTES_PER_WARP {SCRATCH_BYTES_PER_WARP}u
|
||||
"));
|
||||
s.push_str(&format!("// Variant 5: persistent warps, a {} KiB scratch per launched warp (the host launches N warps and passes scratch,\n", p.class.scratch_kb));
|
||||
s.push_str("// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).\n");
|
||||
s.push_str("#define IGNEUM_PERSISTENT_WARPS 1\n");
|
||||
s.push_str(&format!("#define IGNEUM_SCRATCH_OPS {} // scratch read-modify-writes per program ({} per hash)\n", p.class.scratch_slots(), p.scratch_ops_per_hash()));
|
||||
s.push_str(&format!("#define IGNEUM_SCRATCH_SLOTS {}u\n", p.class.scratch_slots_per_lane()));
|
||||
s.push_str(&format!("#define IGNEUM_SCRATCH_WORDS_PER_LANE {}u\n", p.class.scratch_words_per_lane()));
|
||||
s.push_str(&format!("#define IGNEUM_SCRATCH_BYTES_PER_WARP {}u\n", p.class.scratch_bytes_per_warp()));
|
||||
s
|
||||
}
|
||||
|
||||
|
|
@ -472,7 +451,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound:
|
|||
s.push_str(&format!(" constant uint& salt [[buffer({})]],\n", b + 2));
|
||||
s.push_str(" uint tid [[thread_position_in_grid]],\n");
|
||||
s.push_str(" uint nthreads [[threads_per_grid]]) {\n");
|
||||
s.push_str(&persistent_prologue(CoreDialect::Metal));
|
||||
s.push_str(&persistent_prologue(CoreDialect::Metal, p.class.scratch_words_per_lane()));
|
||||
} else {
|
||||
s.push_str(" uint gid [[thread_position_in_grid]]) {\n");
|
||||
}
|
||||
|
|
@ -534,7 +513,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound:
|
|||
}
|
||||
Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&a, false))),
|
||||
Op::WLoad => format!("{d} = {d} ^ {};", fetch(word_index(&a, true))),
|
||||
Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a),
|
||||
Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a, p.class.scratch_slot_mask()),
|
||||
};
|
||||
s.push_str(&format!(" {line} // {k}\n"));
|
||||
}
|
||||
|
|
@ -599,7 +578,7 @@ fn cuda_instr_lines(p: &Program) -> String {
|
|||
Op::Load if load_width(ins) > 1 => wide_load_stmt(CoreDialect::Cuda, &d, &a, ins.width, WideSource::Stored, None),
|
||||
Op::Load => format!("{d} = {d} ^ ds[{a} & mask];"),
|
||||
Op::WLoad => format!("{d} = {d} ^ ds[(__shfl_sync(0xffffffffu, {a}, 0) & wmask) + lane];"),
|
||||
Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a),
|
||||
Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a, p.class.scratch_slot_mask()),
|
||||
};
|
||||
s.push_str(&format!(" {line} // {k} {}\n", ins.op.name()));
|
||||
}
|
||||
|
|
@ -671,7 +650,7 @@ pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String {
|
|||
let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" };
|
||||
s.push_str(&format!("__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask{scratch_args}) {{\n"));
|
||||
if p.has_scratch() {
|
||||
s.push_str(&persistent_prologue(CoreDialect::Cuda));
|
||||
s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane()));
|
||||
} else {
|
||||
s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n");
|
||||
}
|
||||
|
|
@ -792,7 +771,7 @@ pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String {
|
|||
let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" };
|
||||
s.push_str(&format!("__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw{scratch_args}) {{\n"));
|
||||
if p.has_scratch() {
|
||||
s.push_str(&persistent_prologue(CoreDialect::Cuda));
|
||||
s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane()));
|
||||
} else {
|
||||
s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n");
|
||||
}
|
||||
|
|
@ -878,7 +857,7 @@ fn opencl_instr_lines(p: &Program) -> String {
|
|||
Op::Load if load_width(ins) > 1 => wide_load_stmt(CoreDialect::OpenCl, &d, &a, ins.width, WideSource::Stored, None),
|
||||
Op::Load => format!("{d} = {d} ^ ds[{a} & mask];"),
|
||||
Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[(t_ & wmask) + lane]; }}"),
|
||||
Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a),
|
||||
Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a, p.class.scratch_slot_mask()),
|
||||
};
|
||||
s.push_str(&format!(" {line} // {k} {}\n", ins.op.name()));
|
||||
}
|
||||
|
|
@ -897,7 +876,7 @@ pub fn opencl_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String {
|
|||
let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" };
|
||||
s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw{scratch_args}) {{\n"));
|
||||
if p.has_scratch() {
|
||||
s.push_str(&persistent_prologue(CoreDialect::OpenCl));
|
||||
s.push_str(&persistent_prologue(CoreDialect::OpenCl, p.class.scratch_words_per_lane()));
|
||||
} else {
|
||||
s.push_str(" uint gid = (uint)get_global_id(0);\n");
|
||||
}
|
||||
|
|
@ -1029,7 +1008,7 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String {
|
|||
let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" };
|
||||
s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask{scratch_args}) {{\n"));
|
||||
if p.has_scratch() {
|
||||
s.push_str(&persistent_prologue(CoreDialect::OpenCl));
|
||||
s.push_str(&persistent_prologue(CoreDialect::OpenCl, p.class.scratch_words_per_lane()));
|
||||
} else {
|
||||
s.push_str(" uint gid = (uint)get_global_id(0);\n");
|
||||
}
|
||||
|
|
@ -1303,7 +1282,8 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String {
|
|||
s.push_str(&format!(" \"bytes_per_hash\": {},\n", p.bytes_per_hash()));
|
||||
if p.has_scratch() {
|
||||
s.push_str(&format!(" \"scratch_ops_per_hash\": {},\n", p.scratch_ops_per_hash()));
|
||||
s.push_str(&format!(" \"scratch\": \"variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of {SCRATCH_SLOTS} 16-byte slots per lane (lane-major); slot = src & 0x{SCRATCH_SLOT_MASK:x}; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)\",\n"));
|
||||
s.push_str(&format!(" \"scratch_kib_per_warp\": {},\n", p.class.scratch_kb));
|
||||
s.push_str(&format!(" \"scratch\": \"variant 5 (measurement only): persistent warps; a {kb} KiB scratch per warp of {slots} 16-byte slots per lane (lane-major); slot = src & 0x{smask:x}; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)\",\n", kb = p.class.scratch_kb, slots = p.class.scratch_slots_per_lane(), smask = p.class.scratch_slot_mask()));
|
||||
}
|
||||
s.push_str(&format!(" \"wide_load\": \"read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, {FOLD_ROT}) * 0x{FOLD_MUL:08x}) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots\",\n"));
|
||||
}
|
||||
|
|
|
|||
|
|
@ -170,50 +170,70 @@ pub const WIDTH_WORDS: [u8; 3] = [1, 4, 16];
|
|||
pub struct LoadClass {
|
||||
pub mix: [u8; 3],
|
||||
pub load_slots: u8,
|
||||
/// Variant 5: `Some(k)` gives the program a 1 MiB per-warp scratch (the kernels run persistent warps) and
|
||||
/// turns `k` of the load slots into scratch read-modify-writes. `None` for every other class.
|
||||
/// Variant 5: `Some(k)` gives the program a per-warp scratch (the kernels run persistent warps) and turns `k`
|
||||
/// of the load slots into scratch read-modify-writes. `None` for every other class.
|
||||
pub scratch: Option<u8>,
|
||||
/// Variant 5: the scratch per warp in KiB (32 or 128; the whole working set of a card at full occupancy must
|
||||
/// stay under 6 GB, coordinator's cap of 5 October 2026). 0 for every other class.
|
||||
pub scratch_kb: u8,
|
||||
}
|
||||
|
||||
/// Scratch geometry (variant 5): 2^11 slots of 16 bytes per lane (32 KiB), 32 lanes per warp (1 MiB), lane-major.
|
||||
pub const SCRATCH_SLOT_BITS: u32 = 11;
|
||||
pub const SCRATCH_SLOTS: usize = 1 << SCRATCH_SLOT_BITS;
|
||||
pub const SCRATCH_SLOT_MASK: u32 = SCRATCH_SLOTS as u32 - 1;
|
||||
pub const SCRATCH_WORDS_PER_LANE: usize = SCRATCH_SLOTS * 4;
|
||||
pub const SCRATCH_BYTES_PER_WARP: usize = SCRATCH_WORDS_PER_LANE * 4 * LANES;
|
||||
/// Scratch geometry (variant 5): 16-byte slots, lane-major, 32 lanes per warp; `scratch_kb` KiB per warp gives
|
||||
/// `scratch_kb x 2` slots per lane (32 KiB: 64 slots, 128 KiB: 256 slots).
|
||||
pub const SCRATCH_SLOT_BYTES: usize = 16;
|
||||
|
||||
impl LoadClass {
|
||||
/// Slots per lane of the scratch (0 without one).
|
||||
pub fn scratch_slots_per_lane(&self) -> usize {
|
||||
self.scratch_kb as usize * 1024 / LANES / SCRATCH_SLOT_BYTES
|
||||
}
|
||||
pub fn scratch_slot_mask(&self) -> u32 {
|
||||
self.scratch_slots_per_lane().saturating_sub(1) as u32
|
||||
}
|
||||
pub fn scratch_words_per_lane(&self) -> usize {
|
||||
self.scratch_slots_per_lane() * 4
|
||||
}
|
||||
pub fn scratch_bytes_per_warp(&self) -> usize {
|
||||
self.scratch_kb as usize * 1024
|
||||
}
|
||||
}
|
||||
|
||||
impl LoadClass {
|
||||
/// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash.
|
||||
pub const V2: LoadClass = LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None };
|
||||
pub const V2: LoadClass = LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0 };
|
||||
|
||||
/// A fixed width (1, 4 or 16 words) with `load_slots` loads per program.
|
||||
pub fn fixed(width_words: u8, load_slots: u8) -> LoadClass {
|
||||
let mut mix = [0u8; 3];
|
||||
let i = WIDTH_WORDS.iter().position(|&w| w == width_words).expect("width must be 1, 4 or 16 words");
|
||||
mix[i] = 100;
|
||||
LoadClass { mix, load_slots, scratch: None }
|
||||
LoadClass { mix, load_slots, scratch: None, scratch_kb: 0 }
|
||||
}
|
||||
|
||||
/// Per-load width drawn from `mix` (percent for 4, 16, 64 bytes), 16 loads per program.
|
||||
pub fn mixed(mix: [u8; 3]) -> LoadClass {
|
||||
assert_eq!(mix.iter().map(|&m| m as u32).sum::<u32>(), 100, "the mix must sum to 100");
|
||||
LoadClass { mix, load_slots: LOAD_SLOTS as u8, scratch: None }
|
||||
LoadClass { mix, load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0 }
|
||||
}
|
||||
|
||||
/// Variant 5: version 2 widths, 16 memory operations of which `k` are scratch read-modify-writes.
|
||||
pub fn scratch(k: u8) -> LoadClass {
|
||||
/// Variant 5: version 2 widths, 16 memory operations of which `k` are scratch read-modify-writes into a
|
||||
/// scratch of `kb` KiB per warp (a power of two, 1 to 128: at least one slot per lane, under the 6 GB cap).
|
||||
pub fn scratch(k: u8, kb: u8) -> LoadClass {
|
||||
assert!(k as usize <= LOAD_SLOTS);
|
||||
LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: Some(k) }
|
||||
assert!(kb.is_power_of_two() && kb <= 128, "scratch per warp must be a power of two up to 128 KiB");
|
||||
LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: Some(k), scratch_kb: kb }
|
||||
}
|
||||
|
||||
/// Parse "p4,p16,p64" or one of the names of [`LoadClass::name`].
|
||||
/// Parse "p4,p16,p64" or one of the names of [`LoadClass::name`] ("scr4k32": 4 scratch ops, 32 KiB per warp).
|
||||
pub fn parse(s: &str) -> Option<LoadClass> {
|
||||
if let Some(k) = s.strip_prefix("scr") {
|
||||
if let Some(rest) = s.strip_prefix("scr") {
|
||||
let (k, kb) = rest.split_once('k')?;
|
||||
let k: u8 = k.parse().ok()?;
|
||||
if k as usize > LOAD_SLOTS {
|
||||
let kb: u8 = kb.parse().ok()?;
|
||||
if k as usize > LOAD_SLOTS || !kb.is_power_of_two() || kb > 128 {
|
||||
return None;
|
||||
}
|
||||
return Some(LoadClass::scratch(k));
|
||||
return Some(LoadClass::scratch(k, kb));
|
||||
}
|
||||
let (mix_s, slots) = match s.split_once("x") {
|
||||
Some((m, n)) if !m.contains(',') => (m, n.parse::<u8>().ok()?),
|
||||
|
|
@ -235,7 +255,7 @@ impl LoadClass {
|
|||
if slots == 0 || slots as usize >= INSTR_COUNT {
|
||||
return None;
|
||||
}
|
||||
Some(LoadClass { mix, load_slots: slots, scratch: None })
|
||||
Some(LoadClass { mix, load_slots: slots, scratch: None, scratch_kb: 0 })
|
||||
}
|
||||
|
||||
/// Scratch read-modify-writes per program (0 without a scratch).
|
||||
|
|
@ -253,7 +273,7 @@ impl LoadClass {
|
|||
return "v2".to_string();
|
||||
}
|
||||
if let Some(k) = self.scratch {
|
||||
return format!("scr{k}");
|
||||
return format!("scr{k}k{}", self.scratch_kb);
|
||||
}
|
||||
let base = match self.mix {
|
||||
[100, 0, 0] => "w4".to_string(),
|
||||
|
|
@ -387,6 +407,7 @@ pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &L
|
|||
if let Some(k) = class.scratch {
|
||||
b.extend_from_slice(b"scratch/");
|
||||
b.push(k);
|
||||
b.push(class.scratch_kb);
|
||||
}
|
||||
fnv1a64(&b)
|
||||
}
|
||||
|
|
@ -855,10 +876,12 @@ mod tests {
|
|||
assert!(ids.insert(p.program_id()), "{name}: program id collides");
|
||||
}
|
||||
// variant 5: k scratch ops among the 16 memory operations, the rest one-word loads
|
||||
for k in [0u8, 2, 4, 8] {
|
||||
let c = LoadClass::parse(&format!("scr{k}")).unwrap();
|
||||
assert_eq!(c, LoadClass::scratch(k));
|
||||
assert_eq!(c.name(), format!("scr{k}"));
|
||||
for (k, kb) in [(0u8, 32u8), (2, 32), (4, 128), (8, 128)] {
|
||||
let c = LoadClass::parse(&format!("scr{k}k{kb}")).unwrap();
|
||||
assert_eq!(c, LoadClass::scratch(k, kb));
|
||||
assert_eq!(c.name(), format!("scr{k}k{kb}"));
|
||||
assert_eq!(c.scratch_bytes_per_warp(), kb as usize * 1024);
|
||||
assert_eq!(c.scratch_slots_per_lane(), kb as usize * 2);
|
||||
let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c);
|
||||
assert_eq!(p.loads_per_hash(), 128);
|
||||
assert_eq!(p.scratch_ops_per_hash(), 8 * k as usize);
|
||||
|
|
@ -866,7 +889,9 @@ mod tests {
|
|||
assert!(p.instrs.iter().all(|i| i.width == 1));
|
||||
assert!(ids.insert(p.program_id()), "scr{k}: program id collides");
|
||||
}
|
||||
assert!(!LoadClass::scratch(0).is_v2());
|
||||
assert!(!LoadClass::scratch(0, 32).is_v2());
|
||||
assert_ne!(LoadClass::scratch(4, 32).name(), LoadClass::scratch(4, 128).name());
|
||||
assert_eq!(LoadClass::parse("scr4"), None);
|
||||
// a class with the version 2 widths but another slot count takes the extra roll: a different stream
|
||||
let w4x8 = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::fixed(1, 8));
|
||||
assert_ne!(w4x8.instrs, v2.instrs);
|
||||
|
|
|
|||
|
|
@ -1,9 +1,7 @@
|
|||
//! The CPU reference interpreter for one 32-lane warp (`cpuWarpTraced` in the Swift) and the API the node
|
||||
//! calls. Dataset words come from the memory-hard cache (default) or from the closed form (old packs).
|
||||
|
||||
use crate::generator::{
|
||||
generate, generate_class, Instr, LoadClass, Op, Program, ITERATIONS, LANES, SCRATCH_SLOTS, SCRATCH_SLOT_MASK,
|
||||
};
|
||||
use crate::generator::{generate, generate_class, Instr, LoadClass, Op, Program, ITERATIONS, LANES};
|
||||
use crate::memhard::MemhardCpu;
|
||||
use crate::seed::day_key;
|
||||
|
||||
|
|
@ -45,9 +43,10 @@ pub fn scratch_rewrite(x: u32, w: &[u32; 3]) -> [u32; 3] {
|
|||
}
|
||||
|
||||
/// The CPU model of one unit's scratch (variant 5): per lane, the written slots and their words. Unwritten slots
|
||||
/// read as [`scratch_fill`]. A unit touches at most `scratch ops x 32` slots, so the model is small whatever the
|
||||
/// nominal 1 MiB; a GPU keeps the real 1 MiB per resident warp with a per-unit tag per slot.
|
||||
/// read as [`scratch_fill`]. A unit touches at most `scratch ops x 32` slots; a GPU keeps the real scratch per
|
||||
/// resident warp with a per-unit tag per slot.
|
||||
pub struct ScratchModel {
|
||||
slots: usize,
|
||||
written: Vec<bool>,
|
||||
data: Vec<[u32; 3]>,
|
||||
pub reads: usize,
|
||||
|
|
@ -55,13 +54,19 @@ pub struct ScratchModel {
|
|||
}
|
||||
|
||||
impl ScratchModel {
|
||||
pub fn new() -> Self {
|
||||
Self { written: vec![false; LANES * SCRATCH_SLOTS], data: vec![[0; 3]; LANES * SCRATCH_SLOTS], reads: 0, writes: 0 }
|
||||
pub fn new(slots_per_lane: usize) -> Self {
|
||||
Self {
|
||||
slots: slots_per_lane,
|
||||
written: vec![false; LANES * slots_per_lane],
|
||||
data: vec![[0; 3]; LANES * slots_per_lane],
|
||||
reads: 0,
|
||||
writes: 0,
|
||||
}
|
||||
}
|
||||
/// Read slot `slot` of `lane`, then rewrite it from the fold result `x`. Returns the three words read.
|
||||
#[inline]
|
||||
pub fn rmw(&mut self, seed: &[u32; 8], base: u32, lane: usize, slot: u32, dst: u32) -> u32 {
|
||||
let i = lane * SCRATCH_SLOTS + slot as usize;
|
||||
let i = lane * self.slots + slot as usize;
|
||||
let w = if self.written[i] {
|
||||
self.data[i]
|
||||
} else {
|
||||
|
|
@ -80,12 +85,6 @@ impl ScratchModel {
|
|||
}
|
||||
}
|
||||
|
||||
impl Default for ScratchModel {
|
||||
fn default() -> Self {
|
||||
Self::new()
|
||||
}
|
||||
}
|
||||
|
||||
/// Dataset element, closed form of (day words, index). The original prototype's six-operation element.
|
||||
#[inline(always)]
|
||||
pub fn dataset_elem(i: u32, d0: u32, d1: u32) -> u32 {
|
||||
|
|
@ -258,7 +257,8 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32,
|
|||
let mut items_derived = 0usize;
|
||||
let mut idx = [0u32; LANES];
|
||||
let mut val = [0u32; LANES];
|
||||
let mut scratch = if program.has_scratch() { Some(ScratchModel::new()) } else { None };
|
||||
let mut scratch = if program.has_scratch() { Some(ScratchModel::new(program.class.scratch_slots_per_lane())) } else { None };
|
||||
let slot_mask = program.class.scratch_slot_mask();
|
||||
for _ in 0..ITERATIONS {
|
||||
let sel = r[0];
|
||||
for ins in &program.instrs {
|
||||
|
|
@ -267,7 +267,7 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32,
|
|||
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
|
||||
let (d, a) = (ins.dst as usize, ins.src as usize);
|
||||
for lane in 0..LANES {
|
||||
let slot = r[a][lane] & SCRATCH_SLOT_MASK;
|
||||
let slot = r[a][lane] & slot_mask;
|
||||
r[d][lane] = m.rmw(&program.seed, base_nonce, lane, slot, r[d][lane]);
|
||||
}
|
||||
}
|
||||
|
|
@ -538,14 +538,14 @@ mod tests {
|
|||
assert_eq!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 1));
|
||||
assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2));
|
||||
assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 64, 3, 100, 1));
|
||||
let mut m = ScratchModel::new();
|
||||
let mut m = ScratchModel::new(256);
|
||||
let w = [scratch_fill(&seed, 32, 3, 100, 0), scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2)];
|
||||
let x = m.rmw(&seed, 32, 3, 100, 0xabcd);
|
||||
assert_eq!(x, fold_words(0xabcd, &w));
|
||||
let x2 = m.rmw(&seed, 32, 3, 100, 0xabcd);
|
||||
assert_eq!(x2, fold_words(0xabcd, &scratch_rewrite(x, &w)));
|
||||
assert_eq!(m.reads, 2);
|
||||
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::scratch(4));
|
||||
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::scratch(4, 128));
|
||||
assert_eq!(e.program.scratch_ops_per_hash(), 32);
|
||||
assert_eq!(e.hash_warp(0), e.hash_warp(0));
|
||||
}
|
||||
|
|
|
|||
|
|
@ -25,6 +25,9 @@ typedef struct {
|
|||
uint32_t datasetLog2, cacheLog2Words, cacheSegments, datasetMode, generator;
|
||||
uint32_t seedw[8], keyw[8];
|
||||
char seedString[600];
|
||||
// read-width experiment (5 October 2026): the load class (0 when absent), bytes per hash, variant 5's scratch
|
||||
uint32_t loadsPerHash, bytesPerHash, scratchOps, persistent, scratchWordsPerLane;
|
||||
char loadClass[64];
|
||||
// seeds.txt (or program.h): the seeds as the worker protocol carries them
|
||||
char epochHex[65];
|
||||
char dayHex[PF_HEX_CAP];
|
||||
|
|
@ -254,6 +257,12 @@ static int pf_load(const char* dir, PfPack* pk, char* err, size_t cap) {
|
|||
if (!pf_define_u32(prog, "IGNEUM_CACHE_SEGMENTS", &pk->cacheSegments)) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_CACHE_SEGMENTS"); }
|
||||
if (!pf_define_u32(prog, "IGNEUM_GENERATOR", &pk->generator)) pk->generator = 1;
|
||||
if (pf_define_words(prog, "IGNEUM_SEEDW_INIT", pk->seedw, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_SEEDW_INIT with 8 words"); }
|
||||
pk->loadsPerHash = 128; pf_define_u32(prog, "IGNEUM_LOADS_PER_HASH", &pk->loadsPerHash);
|
||||
pk->bytesPerHash = pk->loadsPerHash * 4u; pf_define_u32(prog, "IGNEUM_BYTES_PER_HASH", &pk->bytesPerHash);
|
||||
pk->scratchOps = 0; pf_define_u32(prog, "IGNEUM_SCRATCH_OPS", &pk->scratchOps);
|
||||
pk->persistent = 0; pf_define_u32(prog, "IGNEUM_PERSISTENT_WARPS", &pk->persistent);
|
||||
pk->scratchWordsPerLane = 8192; pf_define_u32(prog, "IGNEUM_SCRATCH_WORDS_PER_LANE", &pk->scratchWordsPerLane);
|
||||
strcpy(pk->loadClass, "v2"); pf_define_str(prog, "IGNEUM_LOAD_CLASS", pk->loadClass, sizeof(pk->loadClass));
|
||||
if (pf_define_words(prog, "IGNEUM_KEY_INIT", pk->keyw, 8) != 8) { free(prog); return pf_fail(err, cap, "program.h has no IGNEUM_KEY_INIT with 8 words"); }
|
||||
if (!pf_define_str(prog, "IGNEUM_SEED_STRING", pk->seedString, sizeof(pk->seedString))) strncpy(pk->seedString, "(no IGNEUM_SEED_STRING)", sizeof(pk->seedString) - 1);
|
||||
pf_define_str(prog, "IGNEUM_SEED_BYTES_HEX", ehex, sizeof(ehex));
|
||||
|
|
|
|||
|
|
@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
|
|||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem
|
|||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
|
|||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
|||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -15,7 +15,7 @@
|
|||
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
|
||||
#define IGNEUM_GENERATOR 2
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0x2f098ae568f38029ull
|
||||
#define IGNEUM_PROGRAM_ID 0xe0b70cd155c28f4bull
|
||||
#define IGNEUM_DAY_STRING "2026-10-03"
|
||||
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
|
||||
#define IGNEUM_DAY0 0x3067619fu
|
||||
|
|
@ -30,20 +30,20 @@
|
|||
#define IGNEUM_OP_MIX "load=16 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 rotl=1"
|
||||
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
|
||||
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
|
||||
#define IGNEUM_LOAD_CLASS "scr0"
|
||||
#define IGNEUM_LOAD_CLASS "scr0k32"
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 16, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 512
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
|
||||
#define IGNEUM_PERSISTENT_WARPS 1
|
||||
#define IGNEUM_SCRATCH_OPS 0 // scratch read-modify-writes per program (0 per hash)
|
||||
#define IGNEUM_SCRATCH_SLOTS 2048u
|
||||
#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u
|
||||
#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u
|
||||
#define IGNEUM_SCRATCH_SLOTS 64u
|
||||
#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u
|
||||
#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
|
|
@ -2,7 +2,7 @@
|
|||
"format": "igneum-program-pack-3",
|
||||
"generator": 2,
|
||||
"attempt": 0,
|
||||
"program_id": "0x2f098ae568f38029",
|
||||
"program_id": "0xe0b70cd155c28f4b",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-genesis",
|
||||
|
|
@ -15,13 +15,14 @@
|
|||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"load_class": "scr0",
|
||||
"load_class": "scr0k32",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [16, 0, 0],
|
||||
"bytes_per_hash": 512,
|
||||
"scratch_ops_per_hash": 0,
|
||||
"scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
|
||||
"scratch_kib_per_warp": 32,
|
||||
"scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"op_mix": {"load": 16, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "rotl": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
|
|
@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
|
|||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
|
|||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
|
|||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
|
|
@ -244,7 +244,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
|
|
@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem
|
|||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -86,7 +86,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
|
|
@ -104,7 +104,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
|
|
@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
|
|||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
|
|
@ -244,7 +244,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
|
|
@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -337,7 +337,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
|
|
@ -355,7 +355,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
|
|
@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
|||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -62,7 +62,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
|
|
@ -80,7 +80,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
67
proto-cuda/packs-readwidth/scr2k128/program.h
Normal file
67
proto-cuda/packs-readwidth/scr2k128/program.h
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
|
||||
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
#include <cuda_runtime.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_SEED_STRING "igneum-genesis"
|
||||
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
|
||||
#define IGNEUM_GENERATOR 2
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0xe0afe0d155bc25d9ull
|
||||
#define IGNEUM_DAY_STRING "2026-10-03"
|
||||
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
|
||||
#define IGNEUM_DAY0 0x3067619fu
|
||||
#define IGNEUM_DAY1 0x3c269176u
|
||||
#define IGNEUM_DATASET_LOG2 28
|
||||
#define IGNEUM_MASK 0x0fffffffu
|
||||
#define IGNEUM_LANES 32
|
||||
#define IGNEUM_ITERATIONS 8
|
||||
#define IGNEUM_INSTR_COUNT 64
|
||||
#define IGNEUM_LOADS_PER_HASH 128
|
||||
#define IGNEUM_WIDE_LOADS_PER_HASH 0
|
||||
#define IGNEUM_OP_MIX "load=14 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 scratch=2 rotl=1"
|
||||
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
|
||||
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
|
||||
#define IGNEUM_LOAD_CLASS "scr2k128"
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 14, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 448
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// Variant 5: persistent warps, a 128 KiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
|
||||
#define IGNEUM_PERSISTENT_WARPS 1
|
||||
#define IGNEUM_SCRATCH_OPS 2 // scratch read-modify-writes per program (16 per hash)
|
||||
#define IGNEUM_SCRATCH_SLOTS 256u
|
||||
#define IGNEUM_SCRATCH_WORDS_PER_LANE 1024u
|
||||
#define IGNEUM_SCRATCH_BYTES_PER_WARP 131072u
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }
|
||||
#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u }
|
||||
#define IGNEUM_CACHE_LOG2_WORDS 26
|
||||
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
|
||||
#define IGNEUM_CACHE_SEGMENTS 65536u
|
||||
#define IGNEUM_ITEM_ROUNDS 8
|
||||
#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u }
|
||||
#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u }
|
||||
#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u }
|
||||
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt);
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#endif
|
||||
|
|
@ -2,7 +2,7 @@
|
|||
"format": "igneum-program-pack-3",
|
||||
"generator": 2,
|
||||
"attempt": 0,
|
||||
"program_id": "0x2f0988e568f37cc3",
|
||||
"program_id": "0xe0afe0d155bc25d9",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-genesis",
|
||||
|
|
@ -15,13 +15,14 @@
|
|||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"load_class": "scr2",
|
||||
"load_class": "scr2k128",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [14, 0, 0],
|
||||
"bytes_per_hash": 448,
|
||||
"scratch_ops_per_hash": 16,
|
||||
"scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
|
||||
"scratch_kib_per_warp": 128,
|
||||
"scratch": "variant 5 (measurement only): persistent warps; a 128 KiB scratch per warp of 256 16-byte slots per lane (lane-major); slot = src & 0xff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"op_mix": {"load": 14, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "scratch": 2, "rotl": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
|
|
@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
|
|||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -70,7 +70,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
|
|
@ -88,7 +88,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
r1 = r1 ^ dataset[r0 & MASK]; // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
r4 = r4 ^ dataset[r0 & MASK]; // 37
|
||||
|
|
@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
|
|||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -72,7 +72,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
|
|
@ -90,7 +90,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
r1 = r1 ^ dataset[r0 & MASK]; // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
r4 = r4 ^ dataset[r0 & MASK]; // 37
|
||||
|
|
@ -11,22 +11,22 @@
|
|||
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
|
||||
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
||||
{ // base nonce 0
|
||||
0x66cffcc97c46e625ull, 0xbc9019f8df50fbfdull, 0x65629c90dde6016eull, 0xd647a41effa03d3bull, 0x86da3b6bbd751b99ull, 0x6ccf4240a0fb2d19ull, 0xeb39a1e06f17378cull, 0x2ea6b349b289fb10ull,
|
||||
0x6067211e6c220500ull, 0x6e6095dfedd1360full, 0xbd1190d8b50e1b48ull, 0x216dc72a0c08d5b5ull, 0x5be1f8c080836b0cull, 0x2a32932a5953ed73ull, 0xcc2a3d68be83c802ull, 0xe46daca15338278full,
|
||||
0xb2de43b96761e459ull, 0x9004acd06588cbeaull, 0x6a9a3543cf93004full, 0xff956d859cb6e408ull, 0x4397ec6e3c5fb045ull, 0x521dea569cd481d5ull, 0x89832b34108759f0ull, 0xf66e393836ffe4eaull,
|
||||
0xb4e39af6c40ea2f4ull, 0x3adc22085dd8d648ull, 0x27efe270958bbfbbull, 0x6c80be0e8dca60d8ull, 0xa0afbc6a60260d59ull, 0x5d9a257fb9189537ull, 0xeb837aeef55dc3edull, 0xd381174dc14f8951ull
|
||||
0x870ae6d97d9e85d8ull, 0x82f91989add778d3ull, 0x127e79aa060861e3ull, 0x148dec51aec6ee46ull, 0xcb5b4b144055bebaull, 0x832ea2d8305f7177ull, 0x374d405f6f35141dull, 0xe83f0b25fcc5b98cull,
|
||||
0x8c6696ba39a3dfbcull, 0x5e588ceed27c0f20ull, 0x87382ec243a8208full, 0x831da1dd50dedfd7ull, 0x11af1b35d85d23faull, 0x3f203e64b5a6ae40ull, 0x5562b30447cf8941ull, 0xdbc5ddc8b06cab3cull,
|
||||
0x089351d365256721ull, 0xf4e67e3e0ce8ce3dull, 0x9a6581aba1e08812ull, 0xf8c1f2017f31d0b2ull, 0x00e282fb6d67ed1bull, 0x64ceb85ec97d5368ull, 0x0db3762c566cd35full, 0x9ecf6fb65b27a141ull,
|
||||
0xdfb6edb29d069ef0ull, 0xa3a2eb24fa67fb93ull, 0x2527bac1b676b544ull, 0x275704b557b5c1d5ull, 0x909fe0fbbb3d7e3eull, 0x6d3b7083f0b9dac1ull, 0xf7a58d446af0c3f0ull, 0x45a10eec5d0361ebull
|
||||
},
|
||||
{ // base nonce 4096
|
||||
0xfb1f61aaeaeaef94ull, 0x184c2160963d8b57ull, 0x42a053c628625778ull, 0xeaad0e41c4770812ull, 0x1d6d389ceb462ce1ull, 0x4a639827672bdbd4ull, 0x857e42aa5a42dd6full, 0xb4ef399e5339979cull,
|
||||
0x497a29225b099233ull, 0x71d8b42862d81954ull, 0x0af995663313bf04ull, 0xf436fd126619d7a1ull, 0x199e4e3333cff269ull, 0x64077952f3775768ull, 0x51af1d126c5e8388ull, 0xffbaf44fe6b15cfdull,
|
||||
0xfc8fed86ecae34a7ull, 0x4cb548616f7a7d6bull, 0xc21d938c8b5bef35ull, 0x34789cbdd7088f71ull, 0xacb099a2c207d891ull, 0xfe1902d162374413ull, 0x26f7831c28f4020bull, 0xdf5192952b4af6b0ull,
|
||||
0xecab61fe88dbaff4ull, 0x941c491f7fdb86e5ull, 0x2b1900c53f746e77ull, 0x8c40507b1caffeb2ull, 0x7532a1ec2b9169efull, 0x1cf399b0c8bfb520ull, 0xdf003d2bb8a2cc0cull, 0x4da853307fc977a9ull
|
||||
0x5ba9a19be2ac506full, 0xf01aba1b9e1fbd4bull, 0x82576a1ada6a06aaull, 0x3cdfb035063961adull, 0x3b1c0146bee5cc0bull, 0xb9eb92e4388bb2edull, 0xfb3d93c5214abf98ull, 0xb624a0986997e24aull,
|
||||
0x6dc096da5e72a34aull, 0x598baa91443c82dbull, 0x689d8cef7afc8df5ull, 0x41ae32225004d576ull, 0x06adbced85e30f4dull, 0x1bef955028e11da8ull, 0x8e3f1fde00391e41ull, 0xf1e29599bd9c776eull,
|
||||
0xd85a42863321dfbcull, 0x2f220e8389179830ull, 0x658f1c559f2e3e28ull, 0x7ddc9adae1172cfaull, 0x493dad4ec6a7d467ull, 0x4f8cf35bbfa01901ull, 0x87971b6666cd0093ull, 0xbe71aa56ca4f56b0ull,
|
||||
0xe9cee6ec93582a96ull, 0xa10e3cc76912027cull, 0x0d3b49bb042a2033ull, 0x680fa00b5d161278ull, 0xdef83e00736c287aull, 0x354ed038aae23286ull, 0xd280ce9bd9a71c97ull, 0x4f48b2b22716cd62ull
|
||||
},
|
||||
{ // base nonce 1000000
|
||||
0x3d094bd04694b96full, 0xaecbd76cecd1a20aull, 0xbcb86febe56b17feull, 0x98082b557ba97517ull, 0xbb5f94108888564bull, 0xea3284877a30fc87ull, 0xc608fa4d5a8bb2adull, 0x946c721c511e0729ull,
|
||||
0x46c6eba292083aedull, 0x936cb97231eb6795ull, 0xb1413c434c712cbbull, 0xedfd554d3948c1bdull, 0xa8a20cbef2faccd5ull, 0x5fe39d756cadbcadull, 0x208b2627380791feull, 0xf52f9374ce480218ull,
|
||||
0xb9db7cd8814eb29eull, 0xf32ed2192b5a8719ull, 0x4f1b06a054940aefull, 0x406df498e4365eb5ull, 0x1982075caad345efull, 0x590f725623dbbbbdull, 0xa26d9192dedfefa5ull, 0x36219ec00da18980ull,
|
||||
0x5361d0dcb0f8b1a3ull, 0x35bdefa2fbb5ffc3ull, 0xba4c2a4e473a9c80ull, 0x107d9d3030f8b9d3ull, 0xa8bb094266d6b987ull, 0x86164fdfbb1426e8ull, 0xa6e8cb895021cbbdull, 0xfe12809e9d99a243ull
|
||||
0x3220aa9dc0bca592ull, 0x5409a7301dfcc3b6ull, 0x31c9b27aad8ef845ull, 0x564ceb4647002ad9ull, 0xcb7dc5fb129b07a6ull, 0xd940b224720c0393ull, 0x0e7f07a345108096ull, 0xc55a7d439b46f6a4ull,
|
||||
0x0a0fab6343d5756aull, 0x397b5e178eeaa82cull, 0xe6a0adaa085838f8ull, 0x7389e3a61a07941bull, 0xeff9f511b46237bbull, 0xb35b6bf8e31ad14cull, 0x0b7d29ad2f411c1bull, 0x988e30319baf21daull,
|
||||
0x362ff74de128b05bull, 0x312357fc7857b317ull, 0xb006adf1d322c446ull, 0x4dcdf54a98b429eeull, 0x18a48dc7b38023bdull, 0xb2af25e04714b6ceull, 0x4d7b97fe2f335dc2ull, 0xc161c509e57b4616ull,
|
||||
0xb2b90c940eb64229ull, 0x2549b669b9e63f78ull, 0x1a1a4ff0e629079cull, 0x80ccafd83d359a54ull, 0x0232c0dfa9240d4dull, 0xc98a751590860b3full, 0x8e96a0aae858e53bull, 0x906b462113b3f109ull
|
||||
}
|
||||
};
|
||||
|
||||
|
|
@ -8,22 +8,22 @@
|
|||
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
|
||||
"warps": [
|
||||
{"base_nonce": 0, "expected": [
|
||||
"0xd0f846c3cd57ae09", "0x4d5e1caf41761a5c", "0x7fc77bb7fa221a51", "0xe45b6317d4f52ae7", "0x09704a39107a3150", "0xd5166620f9dc49ab", "0xeaf1fa69e4075e55", "0x38e23d449b7616ae",
|
||||
"0xf4593424c8427320", "0x9359a44a5149bd60", "0x2df93b5bc274f083", "0xfd80eb32016f8659", "0xdeb10b8a37fc3ee7", "0xc18e2ed77ab77f34", "0x7f39b618cc81c1b4", "0xdc1b7299961ad5ab",
|
||||
"0x8f9c03d151013ec4", "0x668927a4a267d76f", "0x233a50c329caf635", "0x7c80406541ae3f15", "0x18fb38606c48af4a", "0x0452a913df5f112b", "0x59065cb670ba6d7d", "0xa2998e2ccd726df8",
|
||||
"0xef690e221af37893", "0x4e87968e5c45903e", "0x6c6e5d4c1b35d7a9", "0x54d7e322426dcae3", "0x0d3ed6c08deda320", "0x3d6a4301fbb06a5a", "0x9eb78b04d2366566", "0x7806f8c64d2d09ff"
|
||||
"0x870ae6d97d9e85d8", "0x82f91989add778d3", "0x127e79aa060861e3", "0x148dec51aec6ee46", "0xcb5b4b144055beba", "0x832ea2d8305f7177", "0x374d405f6f35141d", "0xe83f0b25fcc5b98c",
|
||||
"0x8c6696ba39a3dfbc", "0x5e588ceed27c0f20", "0x87382ec243a8208f", "0x831da1dd50dedfd7", "0x11af1b35d85d23fa", "0x3f203e64b5a6ae40", "0x5562b30447cf8941", "0xdbc5ddc8b06cab3c",
|
||||
"0x089351d365256721", "0xf4e67e3e0ce8ce3d", "0x9a6581aba1e08812", "0xf8c1f2017f31d0b2", "0x00e282fb6d67ed1b", "0x64ceb85ec97d5368", "0x0db3762c566cd35f", "0x9ecf6fb65b27a141",
|
||||
"0xdfb6edb29d069ef0", "0xa3a2eb24fa67fb93", "0x2527bac1b676b544", "0x275704b557b5c1d5", "0x909fe0fbbb3d7e3e", "0x6d3b7083f0b9dac1", "0xf7a58d446af0c3f0", "0x45a10eec5d0361eb"
|
||||
]},
|
||||
{"base_nonce": 4096, "expected": [
|
||||
"0xe2c97c384a85c687", "0xcf905b005ab01ebc", "0x4526799fae210c6b", "0xd4daf72ed75e8a16", "0xb3703a9e7c6820a8", "0xb6be9393fbb920bd", "0xe2fff04fb7816eb8", "0xed76ad5dbf400f91",
|
||||
"0xea7d4b58ab0eb8d1", "0xf67a73b01f030532", "0x35ab823037510099", "0x5de1b1b0b3a26c69", "0xbe5dd90dbb7e632b", "0x1475918a9237e24d", "0x177fd53c45634d71", "0xa7cf00759ba28ce0",
|
||||
"0xe51be0584ac3fbb4", "0x427049cc778aab35", "0x826bab125577d172", "0xd705b891b16237f5", "0xdf622fd44b180a87", "0x359398ecb79ec2de", "0x2a1e075fb078da66", "0xd480ddd8e66d26e8",
|
||||
"0x07e3e86f517de466", "0xa98b2f7423557445", "0x6bab15b36bb142fe", "0xf87d147bf2cc5c0b", "0xf0294ea2b2820e03", "0xf219ac95e823d794", "0x9fa3fea85bc54264", "0xc4af3fafd2ff5201"
|
||||
"0x5ba9a19be2ac506f", "0xf01aba1b9e1fbd4b", "0x82576a1ada6a06aa", "0x3cdfb035063961ad", "0x3b1c0146bee5cc0b", "0xb9eb92e4388bb2ed", "0xfb3d93c5214abf98", "0xb624a0986997e24a",
|
||||
"0x6dc096da5e72a34a", "0x598baa91443c82db", "0x689d8cef7afc8df5", "0x41ae32225004d576", "0x06adbced85e30f4d", "0x1bef955028e11da8", "0x8e3f1fde00391e41", "0xf1e29599bd9c776e",
|
||||
"0xd85a42863321dfbc", "0x2f220e8389179830", "0x658f1c559f2e3e28", "0x7ddc9adae1172cfa", "0x493dad4ec6a7d467", "0x4f8cf35bbfa01901", "0x87971b6666cd0093", "0xbe71aa56ca4f56b0",
|
||||
"0xe9cee6ec93582a96", "0xa10e3cc76912027c", "0x0d3b49bb042a2033", "0x680fa00b5d161278", "0xdef83e00736c287a", "0x354ed038aae23286", "0xd280ce9bd9a71c97", "0x4f48b2b22716cd62"
|
||||
]},
|
||||
{"base_nonce": 1000000, "expected": [
|
||||
"0x501f772483fac0a3", "0x461363da2c1539e0", "0x750050674de592af", "0x14ed105042cd912c", "0xc6477310878614eb", "0xe916b44e32e90e56", "0x531c92e69c2bdd73", "0x5bb0129ef7c3bd51",
|
||||
"0x967ed91f7cbc7c79", "0x06fcc6a895b58b4a", "0x3bdb29fbf93cbff3", "0xbed385172d2e6abe", "0x921fc99ff4efac5e", "0x6b0090bedd9f69c7", "0x0b105c18aaa53ab0", "0xf0c4203561f3b94d",
|
||||
"0x64abe33adccf7807", "0xd8f3a3b7e0242c04", "0x397da462f69fac5b", "0xe6b6b32e467b71be", "0x5318ac9e56278d04", "0xf7c5e348a4e1f5db", "0x79c994e6109646df", "0x9cdc42b0f6e82231",
|
||||
"0x3f1160f02fd96ac7", "0xa69a23fc7058be21", "0xccde0a19bfdc25b6", "0xd3684a5b966f7497", "0x929db00b97a624f9", "0xfd14a40882d395a6", "0x49b0c1ecc514d6ad", "0x95d994c8349be12d"
|
||||
"0x3220aa9dc0bca592", "0x5409a7301dfcc3b6", "0x31c9b27aad8ef845", "0x564ceb4647002ad9", "0xcb7dc5fb129b07a6", "0xd940b224720c0393", "0x0e7f07a345108096", "0xc55a7d439b46f6a4",
|
||||
"0x0a0fab6343d5756a", "0x397b5e178eeaa82c", "0xe6a0adaa085838f8", "0x7389e3a61a07941b", "0xeff9f511b46237bb", "0xb35b6bf8e31ad14c", "0x0b7d29ad2f411c1b", "0x988e30319baf21da",
|
||||
"0x362ff74de128b05b", "0x312357fc7857b317", "0xb006adf1d322c446", "0x4dcdf54a98b429ee", "0x18a48dc7b38023bd", "0xb2af25e04714b6ce", "0x4d7b97fe2f335dc2", "0xc161c509e57b4616",
|
||||
"0xb2b90c940eb64229", "0x2549b669b9e63f78", "0x1a1a4ff0e629079c", "0x80ccafd83d359a54", "0x0232c0dfa9240d4d", "0xc98a751590860b3f", "0x8e96a0aae858e53b", "0x906b462113b3f109"
|
||||
]}
|
||||
],
|
||||
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],
|
||||
|
|
@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
|
|||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
|
|
@ -242,9 +242,9 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
|
|
@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r1 = r1 ^ ds[r4 & mask]; // 59 load
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
|
|
@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem
|
|||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -86,7 +86,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
|
|
@ -102,9 +102,9 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
|
|
@ -129,7 +129,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
r1 = r1 ^ ds[r4 & mask]; // 59 load
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
|
|
@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
|
|||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -226,7 +226,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
|
|
@ -242,9 +242,9 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
|
|
@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r1 = r1 ^ ds[r4 & mask]; // 59 load
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
|
|
@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -337,7 +337,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
|
|
@ -353,9 +353,9 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
|
|
@ -380,7 +380,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r1 = r1 ^ ds[r4 & mask]; // 59 load
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
|
|
@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
|||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -62,7 +62,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
|
|
@ -78,9 +78,9 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
r1 = r1 ^ ds[r0 & mask]; // 32 load
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
|
|
@ -105,7 +105,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
r1 = r1 ^ ds[r4 & mask]; // 59 load
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
|
|
@ -15,7 +15,7 @@
|
|||
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
|
||||
#define IGNEUM_GENERATOR 2
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0x2f0988e568f37cc3ull
|
||||
#define IGNEUM_PROGRAM_ID 0xe0b080d155bd35b9ull
|
||||
#define IGNEUM_DAY_STRING "2026-10-03"
|
||||
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
|
||||
#define IGNEUM_DAY0 0x3067619fu
|
||||
|
|
@ -30,20 +30,20 @@
|
|||
#define IGNEUM_OP_MIX "load=14 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 scratch=2 rotl=1"
|
||||
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
|
||||
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
|
||||
#define IGNEUM_LOAD_CLASS "scr2"
|
||||
#define IGNEUM_LOAD_CLASS "scr2k32"
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 14, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 448
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
|
||||
#define IGNEUM_PERSISTENT_WARPS 1
|
||||
#define IGNEUM_SCRATCH_OPS 2 // scratch read-modify-writes per program (16 per hash)
|
||||
#define IGNEUM_SCRATCH_SLOTS 2048u
|
||||
#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u
|
||||
#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u
|
||||
#define IGNEUM_SCRATCH_SLOTS 64u
|
||||
#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u
|
||||
#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
130
proto-cuda/packs-readwidth/scr2k32/program.json
Normal file
130
proto-cuda/packs-readwidth/scr2k32/program.json
Normal file
|
|
@ -0,0 +1,130 @@
|
|||
{
|
||||
"format": "igneum-program-pack-3",
|
||||
"generator": 2,
|
||||
"attempt": 0,
|
||||
"program_id": "0xe0b080d155bd35b9",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-genesis",
|
||||
"seed_bytes": "69676e65756d2d67656e65736973",
|
||||
"seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"],
|
||||
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
|
||||
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
|
||||
"lanes": 32,
|
||||
"registers": 8,
|
||||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"load_class": "scr2k32",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [14, 0, 0],
|
||||
"bytes_per_hash": 448,
|
||||
"scratch_ops_per_hash": 16,
|
||||
"scratch_kib_per_warp": 32,
|
||||
"scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"op_mix": {"load": 14, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "scratch": 2, "rotl": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
|
||||
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
|
||||
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
|
||||
"op_semantics": {
|
||||
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
|
||||
"sub": "dst = dst - src",
|
||||
"mul": "dst = dst * src (low 32)",
|
||||
"mulhi": "dst = high 32 bits of dst * src",
|
||||
"xor": "dst = dst ^ src",
|
||||
"or": "dst = dst | src",
|
||||
"rotl": "dst = rotl(dst, rot), rot in 1..31",
|
||||
"rotr": "dst = rotr(dst, src & 31)",
|
||||
"mad": "dst = src * src2 + dst",
|
||||
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
|
||||
"load": "dst = dst ^ dataset[src & dataset.mask]",
|
||||
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
|
||||
},
|
||||
"dataset": {
|
||||
"log2_words": 28,
|
||||
"bytes": 1073741824,
|
||||
"mask": "0x0fffffff",
|
||||
"day": "2026-10-03",
|
||||
"day_bytes": "6461792f323032362d31302d3033",
|
||||
"day_words_from": "seed_words_from_bytes(day_bytes)",
|
||||
"d0": "0x3067619f",
|
||||
"d1": "0x3c269176",
|
||||
"mode": "memory-hard",
|
||||
"spec": "proto-metal/MEMHARD.md",
|
||||
"key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"],
|
||||
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
|
||||
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
|
||||
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
|
||||
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s",
|
||||
"word": "dataset[w] = item(w >> 4)[w & 15]"
|
||||
},
|
||||
"instructions": [
|
||||
{"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1},
|
||||
{"i": 1, "op": "add", "dst": 1, "src": 7, "src2": 2, "imm": "0x42da7657", "imm2": "0xc3bd2355", "rot": 25, "bit": 4, "mask": 16, "width": 1},
|
||||
{"i": 2, "op": "add", "dst": 2, "src": 3, "src2": 0, "imm": "0x61f0b51c", "imm2": "0x2735a174", "rot": 4, "bit": 26, "mask": 2, "width": 1},
|
||||
{"i": 3, "op": "mad", "dst": 4, "src": 0, "src2": 6, "imm": "0x679648a8", "imm2": "0x3044ba32", "rot": 31, "bit": 31, "mask": 4, "width": 1},
|
||||
{"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1},
|
||||
{"i": 5, "op": "load", "dst": 4, "src": 1, "src2": 2, "imm": "0x987c017a", "imm2": "0xf4d60559", "rot": 2, "bit": 0, "mask": 4, "width": 1},
|
||||
{"i": 6, "op": "shfl", "dst": 6, "src": 3, "src2": 7, "imm": "0x6ea7b2df", "imm2": "0x9fce5071", "rot": 7, "bit": 15, "mask": 4, "width": 1},
|
||||
{"i": 7, "op": "shfl", "dst": 1, "src": 5, "src2": 1, "imm": "0x26a2ecde", "imm2": "0xfec6ad22", "rot": 15, "bit": 11, "mask": 8, "width": 1},
|
||||
{"i": 8, "op": "xor", "dst": 7, "src": 5, "src2": 2, "imm": "0xbe4b445c", "imm2": "0x17a5a9c7", "rot": 8, "bit": 8, "mask": 1, "width": 1},
|
||||
{"i": 9, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1},
|
||||
{"i": 10, "op": "or", "dst": 1, "src": 2, "src2": 3, "imm": "0x4e7dc10d", "imm2": "0x196d165c", "rot": 14, "bit": 27, "mask": 16, "width": 1},
|
||||
{"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 1, "imm": "0xc5c3b55d", "imm2": "0xec061424", "rot": 26, "bit": 27, "mask": 8, "width": 1},
|
||||
{"i": 12, "op": "or", "dst": 6, "src": 2, "src2": 3, "imm": "0x306542fe", "imm2": "0x1bb1b429", "rot": 31, "bit": 0, "mask": 2, "width": 1},
|
||||
{"i": 13, "op": "mul", "dst": 2, "src": 5, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1},
|
||||
{"i": 14, "op": "load", "dst": 1, "src": 2, "src2": 5, "imm": "0x028b4d37", "imm2": "0x7bbd78ea", "rot": 15, "bit": 2, "mask": 8, "width": 1},
|
||||
{"i": 15, "op": "rotl", "dst": 7, "src": 6, "src2": 6, "imm": "0x5c88a1a7", "imm2": "0x5c628769", "rot": 1, "bit": 3, "mask": 8, "width": 1},
|
||||
{"i": 16, "op": "scratch", "dst": 3, "src": 6, "src2": 7, "imm": "0xbac2ae81", "imm2": "0xcbbc7bdb", "rot": 18, "bit": 8, "mask": 8, "width": 1},
|
||||
{"i": 17, "op": "load", "dst": 7, "src": 4, "src2": 2, "imm": "0xe8ab93e9", "imm2": "0xa00de107", "rot": 2, "bit": 1, "mask": 16, "width": 1},
|
||||
{"i": 18, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1},
|
||||
{"i": 19, "op": "mad", "dst": 4, "src": 0, "src2": 2, "imm": "0x5fba7bc2", "imm2": "0xdf099cfb", "rot": 4, "bit": 15, "mask": 16, "width": 1},
|
||||
{"i": 20, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0x0a3056de", "imm2": "0x7f0c25c3", "rot": 27, "bit": 13, "mask": 8, "width": 1},
|
||||
{"i": 21, "op": "xor", "dst": 5, "src": 7, "src2": 4, "imm": "0xbd066e1d", "imm2": "0x6d3ddc5a", "rot": 2, "bit": 29, "mask": 1, "width": 1},
|
||||
{"i": 22, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1},
|
||||
{"i": 23, "op": "load", "dst": 3, "src": 7, "src2": 2, "imm": "0xc7fcfc8f", "imm2": "0x8528b94f", "rot": 17, "bit": 13, "mask": 4, "width": 1},
|
||||
{"i": 24, "op": "mulhi", "dst": 7, "src": 3, "src2": 5, "imm": "0xd91641e8", "imm2": "0xaf77faf2", "rot": 22, "bit": 21, "mask": 1, "width": 1},
|
||||
{"i": 25, "op": "or", "dst": 5, "src": 4, "src2": 0, "imm": "0x84c03868", "imm2": "0xf6c691b7", "rot": 29, "bit": 14, "mask": 8, "width": 1},
|
||||
{"i": 26, "op": "mad", "dst": 4, "src": 5, "src2": 2, "imm": "0x3bb2b6ba", "imm2": "0x49d95fd5", "rot": 1, "bit": 5, "mask": 8, "width": 1},
|
||||
{"i": 27, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1},
|
||||
{"i": 28, "op": "mulhi", "dst": 6, "src": 7, "src2": 6, "imm": "0xd69c4715", "imm2": "0xe0ebc4ce", "rot": 29, "bit": 2, "mask": 8, "width": 1},
|
||||
{"i": 29, "op": "add", "dst": 6, "src": 1, "src2": 2, "imm": "0x3b2d2124", "imm2": "0x187a9128", "rot": 1, "bit": 9, "mask": 16, "width": 1},
|
||||
{"i": 30, "op": "rotr", "dst": 6, "src": 7, "src2": 0, "imm": "0x5c64a589", "imm2": "0x61c9a38d", "rot": 17, "bit": 21, "mask": 16, "width": 1},
|
||||
{"i": 31, "op": "load", "dst": 3, "src": 1, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1},
|
||||
{"i": 32, "op": "load", "dst": 1, "src": 0, "src2": 7, "imm": "0xcc7972c4", "imm2": "0xad098d15", "rot": 30, "bit": 21, "mask": 8, "width": 1},
|
||||
{"i": 33, "op": "add", "dst": 0, "src": 4, "src2": 4, "imm": "0x2c35699f", "imm2": "0x351dde38", "rot": 21, "bit": 18, "mask": 4, "width": 1},
|
||||
{"i": 34, "op": "scratch", "dst": 0, "src": 2, "src2": 3, "imm": "0xfae8902b", "imm2": "0x5cd8306f", "rot": 5, "bit": 28, "mask": 16, "width": 1},
|
||||
{"i": 35, "op": "mul", "dst": 0, "src": 3, "src2": 1, "imm": "0x4fa3f3db", "imm2": "0xdbf37e75", "rot": 7, "bit": 18, "mask": 4, "width": 1},
|
||||
{"i": 36, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1},
|
||||
{"i": 37, "op": "load", "dst": 4, "src": 0, "src2": 0, "imm": "0x04cc1d55", "imm2": "0x35c52d04", "rot": 11, "bit": 14, "mask": 2, "width": 1},
|
||||
{"i": 38, "op": "mad", "dst": 1, "src": 3, "src2": 5, "imm": "0x3958f280", "imm2": "0x8713c7e1", "rot": 5, "bit": 23, "mask": 16, "width": 1},
|
||||
{"i": 39, "op": "add", "dst": 0, "src": 3, "src2": 3, "imm": "0xa907b90b", "imm2": "0x1b053acf", "rot": 30, "bit": 25, "mask": 16, "width": 1},
|
||||
{"i": 40, "op": "rotr", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1},
|
||||
{"i": 41, "op": "mul", "dst": 3, "src": 2, "src2": 4, "imm": "0x49087d74", "imm2": "0x6348b489", "rot": 17, "bit": 9, "mask": 16, "width": 1},
|
||||
{"i": 42, "op": "add", "dst": 1, "src": 5, "src2": 1, "imm": "0xa32e000c", "imm2": "0x6058c2e3", "rot": 25, "bit": 20, "mask": 8, "width": 1},
|
||||
{"i": 43, "op": "xor", "dst": 3, "src": 4, "src2": 2, "imm": "0x3dad0eb6", "imm2": "0xb97578cb", "rot": 3, "bit": 27, "mask": 1, "width": 1},
|
||||
{"i": 44, "op": "load", "dst": 3, "src": 5, "src2": 7, "imm": "0x374aec92", "imm2": "0x626f11df", "rot": 20, "bit": 18, "mask": 8, "width": 1},
|
||||
{"i": 45, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1},
|
||||
{"i": 46, "op": "xor", "dst": 7, "src": 1, "src2": 0, "imm": "0xef6ac348", "imm2": "0x963bb7e6", "rot": 26, "bit": 3, "mask": 8, "width": 1},
|
||||
{"i": 47, "op": "add", "dst": 0, "src": 3, "src2": 0, "imm": "0x838b5065", "imm2": "0x36360066", "rot": 3, "bit": 31, "mask": 4, "width": 1},
|
||||
{"i": 48, "op": "mulhi", "dst": 7, "src": 5, "src2": 0, "imm": "0x8458f7ac", "imm2": "0xc1c15026", "rot": 27, "bit": 15, "mask": 8, "width": 1},
|
||||
{"i": 49, "op": "load", "dst": 0, "src": 2, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1},
|
||||
{"i": 50, "op": "sub", "dst": 2, "src": 6, "src2": 0, "imm": "0x2baec8c9", "imm2": "0x4390f156", "rot": 3, "bit": 12, "mask": 8, "width": 1},
|
||||
{"i": 51, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x19234061", "imm2": "0xe84dfade", "rot": 4, "bit": 19, "mask": 1, "width": 1},
|
||||
{"i": 52, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0xdc2cd71e", "imm2": "0x1b5d334b", "rot": 9, "bit": 8, "mask": 8, "width": 1},
|
||||
{"i": 53, "op": "sub", "dst": 7, "src": 0, "src2": 4, "imm": "0x605c31ec", "imm2": "0x9923ff88", "rot": 28, "bit": 25, "mask": 4, "width": 1},
|
||||
{"i": 54, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1},
|
||||
{"i": 55, "op": "xor", "dst": 7, "src": 5, "src2": 5, "imm": "0xad7493e7", "imm2": "0x3e400372", "rot": 13, "bit": 8, "mask": 1, "width": 1},
|
||||
{"i": 56, "op": "load", "dst": 2, "src": 7, "src2": 1, "imm": "0x87e933c9", "imm2": "0x8c854c1b", "rot": 17, "bit": 3, "mask": 8, "width": 1},
|
||||
{"i": 57, "op": "sub", "dst": 5, "src": 6, "src2": 5, "imm": "0x11be3bc9", "imm2": "0xbbaa8e24", "rot": 6, "bit": 5, "mask": 16, "width": 1},
|
||||
{"i": 58, "op": "load", "dst": 1, "src": 3, "src2": 2, "imm": "0xa732351a", "imm2": "0xc01349cd", "rot": 14, "bit": 17, "mask": 16, "width": 1},
|
||||
{"i": 59, "op": "load", "dst": 1, "src": 4, "src2": 0, "imm": "0xb20547b2", "imm2": "0xc94655de", "rot": 27, "bit": 30, "mask": 1, "width": 1},
|
||||
{"i": 60, "op": "sub", "dst": 4, "src": 6, "src2": 7, "imm": "0x67cf904c", "imm2": "0x6873b216", "rot": 27, "bit": 7, "mask": 16, "width": 1},
|
||||
{"i": 61, "op": "mul", "dst": 1, "src": 2, "src2": 7, "imm": "0x93ab0bf4", "imm2": "0x96158375", "rot": 14, "bit": 0, "mask": 16, "width": 1},
|
||||
{"i": 62, "op": "mad", "dst": 3, "src": 6, "src2": 0, "imm": "0x41a443a3", "imm2": "0xe69d7919", "rot": 9, "bit": 0, "mask": 16, "width": 1},
|
||||
{"i": 63, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x2fe0e98b", "imm2": "0xc88e2942", "rot": 5, "bit": 16, "mask": 16, "width": 1}
|
||||
]
|
||||
}
|
||||
|
|
@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
|
|||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -70,7 +70,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
|
|
@ -86,9 +86,9 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
|
||||
r6 = rotr_var(r6, r7); // 30
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
r1 = r1 ^ dataset[r0 & MASK]; // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
r4 = r4 ^ dataset[r0 & MASK]; // 37
|
||||
|
|
@ -113,7 +113,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
r2 = r2 ^ dataset[r7 & MASK]; // 56
|
||||
r5 = r5 - r6; // 57
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 58
|
||||
{ uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
r1 = r1 ^ dataset[r4 & MASK]; // 59
|
||||
r4 = r4 - r6; // 60
|
||||
r1 = r1 * r2; // 61
|
||||
r3 = r6 * r0 + r3; // 62
|
||||
|
|
@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
|
|||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -72,7 +72,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
|
|
@ -88,9 +88,9 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
|
||||
r6 = rotr_var(r6, r7); // 30
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
r1 = r1 ^ dataset[r0 & MASK]; // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
r4 = r4 ^ dataset[r0 & MASK]; // 37
|
||||
|
|
@ -115,7 +115,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
r2 = r2 ^ dataset[r7 & MASK]; // 56
|
||||
r5 = r5 - r6; // 57
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 58
|
||||
{ uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
r1 = r1 ^ dataset[r4 & MASK]; // 59
|
||||
r4 = r4 - r6; // 60
|
||||
r1 = r1 * r2; // 61
|
||||
r3 = r6 * r0 + r3; // 62
|
||||
|
|
@ -11,22 +11,22 @@
|
|||
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
|
||||
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
||||
{ // base nonce 0
|
||||
0xd0f846c3cd57ae09ull, 0x4d5e1caf41761a5cull, 0x7fc77bb7fa221a51ull, 0xe45b6317d4f52ae7ull, 0x09704a39107a3150ull, 0xd5166620f9dc49abull, 0xeaf1fa69e4075e55ull, 0x38e23d449b7616aeull,
|
||||
0xf4593424c8427320ull, 0x9359a44a5149bd60ull, 0x2df93b5bc274f083ull, 0xfd80eb32016f8659ull, 0xdeb10b8a37fc3ee7ull, 0xc18e2ed77ab77f34ull, 0x7f39b618cc81c1b4ull, 0xdc1b7299961ad5abull,
|
||||
0x8f9c03d151013ec4ull, 0x668927a4a267d76full, 0x233a50c329caf635ull, 0x7c80406541ae3f15ull, 0x18fb38606c48af4aull, 0x0452a913df5f112bull, 0x59065cb670ba6d7dull, 0xa2998e2ccd726df8ull,
|
||||
0xef690e221af37893ull, 0x4e87968e5c45903eull, 0x6c6e5d4c1b35d7a9ull, 0x54d7e322426dcae3ull, 0x0d3ed6c08deda320ull, 0x3d6a4301fbb06a5aull, 0x9eb78b04d2366566ull, 0x7806f8c64d2d09ffull
|
||||
0x589f62cd61c27dc1ull, 0xf6be7bab8b00a7c7ull, 0x349551c5979e330eull, 0x6d8156f5afaf0064ull, 0x81971552315d55b8ull, 0x20a3bb31ef5e202cull, 0x91818c4ec9fd5fecull, 0x35ca8cde74ac9715ull,
|
||||
0xc1db0a90f80c8801ull, 0xbc52521a33053a8aull, 0x89fda590bf946dd6ull, 0xc54fe4aeaf11975full, 0xffc712960d7f4022ull, 0x4da6b3de6f9a0abdull, 0xf70ff0468e9b595aull, 0xc2d1d4434c1eb8c9ull,
|
||||
0x70dc278b36920857ull, 0xfb20a2d85fa65b83ull, 0xd24316c07937542dull, 0xd505e1e3e85694bdull, 0x88c759248ae122a7ull, 0xf1c04d8ad71d5e8cull, 0x7e4d0e4e99717f12ull, 0x63d9e4b8f619ac98ull,
|
||||
0xcfef2a4243ba8608ull, 0xb2908c3da3593ba9ull, 0x48bee70b2ee45b84ull, 0x8f7a09d71be2c321ull, 0x0aadd48107e05bdbull, 0x8b73ff0f34ddaaaaull, 0x9ff3a4875ecf0fe3ull, 0xaf3d701cd1cbbd4bull
|
||||
},
|
||||
{ // base nonce 4096
|
||||
0xe2c97c384a85c687ull, 0xcf905b005ab01ebcull, 0x4526799fae210c6bull, 0xd4daf72ed75e8a16ull, 0xb3703a9e7c6820a8ull, 0xb6be9393fbb920bdull, 0xe2fff04fb7816eb8ull, 0xed76ad5dbf400f91ull,
|
||||
0xea7d4b58ab0eb8d1ull, 0xf67a73b01f030532ull, 0x35ab823037510099ull, 0x5de1b1b0b3a26c69ull, 0xbe5dd90dbb7e632bull, 0x1475918a9237e24dull, 0x177fd53c45634d71ull, 0xa7cf00759ba28ce0ull,
|
||||
0xe51be0584ac3fbb4ull, 0x427049cc778aab35ull, 0x826bab125577d172ull, 0xd705b891b16237f5ull, 0xdf622fd44b180a87ull, 0x359398ecb79ec2deull, 0x2a1e075fb078da66ull, 0xd480ddd8e66d26e8ull,
|
||||
0x07e3e86f517de466ull, 0xa98b2f7423557445ull, 0x6bab15b36bb142feull, 0xf87d147bf2cc5c0bull, 0xf0294ea2b2820e03ull, 0xf219ac95e823d794ull, 0x9fa3fea85bc54264ull, 0xc4af3fafd2ff5201ull
|
||||
0xc93652d639480287ull, 0x2bff33a7119c0ec9ull, 0x7c676a1cf9f37474ull, 0xcb1dab3b216bdd52ull, 0x236552ac169e0f4aull, 0xffb7303a10cb2833ull, 0xdf1ca9f83cf0740cull, 0xe88fcc23bd9a0a9full,
|
||||
0x48118af81df77459ull, 0x49379fea0c36ec78ull, 0x931e8c0ca930cd35ull, 0xba3e0b2487710abfull, 0x56b607c6f3672398ull, 0xeb0215d7735482c3ull, 0x104b2405ae428d28ull, 0xa8c21e4eb7e2b744ull,
|
||||
0xbc41f10ffa4e4621ull, 0xc51ab216e6e0a339ull, 0x285cdde4bd22e712ull, 0x4f985d36d1302ebdull, 0x82563e5cd9a31b86ull, 0x9495c481dd399662ull, 0xec6111e88d79f207ull, 0x112bf6a954166121ull,
|
||||
0x1a6eda3d1a3846b6ull, 0xd25ebd17cbfaf07dull, 0xe552179181d4390full, 0xdd0129cf2d8db153ull, 0x0863f60becfbabedull, 0x1825c21e13698cecull, 0x20acd589e0408f6eull, 0x817a8413e24a68d4ull
|
||||
},
|
||||
{ // base nonce 1000000
|
||||
0x501f772483fac0a3ull, 0x461363da2c1539e0ull, 0x750050674de592afull, 0x14ed105042cd912cull, 0xc6477310878614ebull, 0xe916b44e32e90e56ull, 0x531c92e69c2bdd73ull, 0x5bb0129ef7c3bd51ull,
|
||||
0x967ed91f7cbc7c79ull, 0x06fcc6a895b58b4aull, 0x3bdb29fbf93cbff3ull, 0xbed385172d2e6abeull, 0x921fc99ff4efac5eull, 0x6b0090bedd9f69c7ull, 0x0b105c18aaa53ab0ull, 0xf0c4203561f3b94dull,
|
||||
0x64abe33adccf7807ull, 0xd8f3a3b7e0242c04ull, 0x397da462f69fac5bull, 0xe6b6b32e467b71beull, 0x5318ac9e56278d04ull, 0xf7c5e348a4e1f5dbull, 0x79c994e6109646dfull, 0x9cdc42b0f6e82231ull,
|
||||
0x3f1160f02fd96ac7ull, 0xa69a23fc7058be21ull, 0xccde0a19bfdc25b6ull, 0xd3684a5b966f7497ull, 0x929db00b97a624f9ull, 0xfd14a40882d395a6ull, 0x49b0c1ecc514d6adull, 0x95d994c8349be12dull
|
||||
0x7535b29ea3e2823eull, 0xd84fe0e0281b6538ull, 0x14f4b14bf5185a26ull, 0x5fc4ec481d8c65bcull, 0x9ff6a4ec626c4cbcull, 0x52131ecd506117f0ull, 0x9a7db1822213f9b9ull, 0x025ac827f2f88c7cull,
|
||||
0x4076da8ca02131e1ull, 0xbc9956bc70d53e0bull, 0x077cf7d297357750ull, 0xb00c8db428fbccf2ull, 0x50f16eecd1fb65c4ull, 0x4daddb3cf455583dull, 0x952b3cca95e87c92ull, 0xa7b7af6eac1a0222ull,
|
||||
0xc58c0db5a99ced05ull, 0xa71a8ce697d65e94ull, 0xe29bab54459076d2ull, 0x5f613619a76c6400ull, 0xb43e9559e242a8d4ull, 0x4b5433e68aa1f302ull, 0x382b7f105840032cull, 0xbf402649fb8e9618ull,
|
||||
0xf45994bb29726a41ull, 0x8a11f358c24795cbull, 0xc2c4f8902007527cull, 0xe12f65396a832dcdull, 0x307d3f495790aff0ull, 0x5fc00c5eb0c3e81eull, 0xb46600e4685191eeull, 0xce65db0e1d36f875ull
|
||||
}
|
||||
};
|
||||
|
||||
|
|
@ -8,22 +8,22 @@
|
|||
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
|
||||
"warps": [
|
||||
{"base_nonce": 0, "expected": [
|
||||
"0x2273e2732203e32a", "0xa513d354bd107990", "0xe005b7515054c85f", "0x18a61b37b30cd1fb", "0xa21d5b98e8d07e9c", "0x5c24171a391d5ed0", "0x0f29e7583e1794b8", "0x7eca8a374d1a4f70",
|
||||
"0x260ca011cd9ea10c", "0xce1050798fce3d43", "0x9560d939dca19041", "0x8480a8440b80ecc3", "0xfaa99aac459b739e", "0x7f083e72458e08ab", "0x78d876842f68672b", "0x3b9bcf6275d3575c",
|
||||
"0x0256af61bdbf11b3", "0xefc6771cae646cbd", "0xbc44f1c9f9f54d87", "0x6caedd783487eb7d", "0x001b31fcb4fe0d4d", "0x947a7ba1057e25b6", "0xb9e5a0204d68c22a", "0x50489bed25d42661",
|
||||
"0xb1019bff6d1057cd", "0xd1442990562ce940", "0xcd986a47f98801db", "0x9c8796b6df23300f", "0xbd53ad05d2c877a9", "0xc95e863774a15b0a", "0x132d8a91fb2fa67a", "0x53dd38e8eadc8a24"
|
||||
"0x589f62cd61c27dc1", "0xf6be7bab8b00a7c7", "0x349551c5979e330e", "0x6d8156f5afaf0064", "0x81971552315d55b8", "0x20a3bb31ef5e202c", "0x91818c4ec9fd5fec", "0x35ca8cde74ac9715",
|
||||
"0xc1db0a90f80c8801", "0xbc52521a33053a8a", "0x89fda590bf946dd6", "0xc54fe4aeaf11975f", "0xffc712960d7f4022", "0x4da6b3de6f9a0abd", "0xf70ff0468e9b595a", "0xc2d1d4434c1eb8c9",
|
||||
"0x70dc278b36920857", "0xfb20a2d85fa65b83", "0xd24316c07937542d", "0xd505e1e3e85694bd", "0x88c759248ae122a7", "0xf1c04d8ad71d5e8c", "0x7e4d0e4e99717f12", "0x63d9e4b8f619ac98",
|
||||
"0xcfef2a4243ba8608", "0xb2908c3da3593ba9", "0x48bee70b2ee45b84", "0x8f7a09d71be2c321", "0x0aadd48107e05bdb", "0x8b73ff0f34ddaaaa", "0x9ff3a4875ecf0fe3", "0xaf3d701cd1cbbd4b"
|
||||
]},
|
||||
{"base_nonce": 4096, "expected": [
|
||||
"0x78c93312a03fb0ee", "0x184fea638ec9b5fb", "0x5687e8dcc4301dbf", "0xed02c94f23681dfc", "0x326d70162241ff6d", "0x452017eb4ed2dfcf", "0xc10b0e016f1e28c9", "0x691ce0cecf2a99ba",
|
||||
"0x6c9506f34e0e63ce", "0x447a98c2b7fdfa40", "0x07486b0e4055b2c9", "0x41781460bd47fd5c", "0x01db316e35198291", "0xccd7e727f139a880", "0xdd7bd9efd16bf21c", "0x8285d37966656366",
|
||||
"0x383deade15fe0ecb", "0x5fd64f5873c8e324", "0xad584cb6839c5e1d", "0xbb842707fb5e9460", "0x4e8bc8f87978fcbd", "0x18eb56f4a1fae881", "0x4c3b731a6b0c47a1", "0xda52cf9d69b252eb",
|
||||
"0xb5ff19b2b3eeb13e", "0xe2595cea2afe42dd", "0x3ff108424c9e6e38", "0x3a8a9e1995f359ca", "0x6a6b1da662cf2126", "0x54e684c127bb181f", "0x2018caa81f1a7d50", "0x33e94d2c92d9d148"
|
||||
"0xc93652d639480287", "0x2bff33a7119c0ec9", "0x7c676a1cf9f37474", "0xcb1dab3b216bdd52", "0x236552ac169e0f4a", "0xffb7303a10cb2833", "0xdf1ca9f83cf0740c", "0xe88fcc23bd9a0a9f",
|
||||
"0x48118af81df77459", "0x49379fea0c36ec78", "0x931e8c0ca930cd35", "0xba3e0b2487710abf", "0x56b607c6f3672398", "0xeb0215d7735482c3", "0x104b2405ae428d28", "0xa8c21e4eb7e2b744",
|
||||
"0xbc41f10ffa4e4621", "0xc51ab216e6e0a339", "0x285cdde4bd22e712", "0x4f985d36d1302ebd", "0x82563e5cd9a31b86", "0x9495c481dd399662", "0xec6111e88d79f207", "0x112bf6a954166121",
|
||||
"0x1a6eda3d1a3846b6", "0xd25ebd17cbfaf07d", "0xe552179181d4390f", "0xdd0129cf2d8db153", "0x0863f60becfbabed", "0x1825c21e13698cec", "0x20acd589e0408f6e", "0x817a8413e24a68d4"
|
||||
]},
|
||||
{"base_nonce": 1000000, "expected": [
|
||||
"0x04a41389bf3dfd3d", "0xc509164def9207df", "0x4a8ffdbdf46e429d", "0xff13bf0dc1b39aeb", "0xb852acc8e24133d7", "0x4bdd991ae56252ac", "0xa7739e74b3a054e9", "0xb4e36218d4b45fdc",
|
||||
"0x8bbd323155f5edc5", "0xb7b56a90659e7fd2", "0xdff7c495b7027480", "0xffa8adb5c0302b06", "0xe97d7967d89a5672", "0x0d0c2d4e6493926e", "0xe9a5cda333cf2043", "0xdc95256d0986e5d8",
|
||||
"0xd0dc211b811d6843", "0x68dfa3d0fb9a569b", "0xa9e0028dfd9178c0", "0x4a36ca1fc40b20a9", "0xe7c765c5a735294b", "0xf08954b015cb2628", "0xc69ee66ecf2740c5", "0xe3d01e899e46b089",
|
||||
"0xc3558c74159c8603", "0x4c7aeb196bd01b04", "0x13c17119385f1910", "0xda7fca98e0989b8a", "0x6f95baf340817945", "0x1af52756fd3afcab", "0xb8eefc370bbe7e4b", "0x94a55055b48bd4db"
|
||||
"0x7535b29ea3e2823e", "0xd84fe0e0281b6538", "0x14f4b14bf5185a26", "0x5fc4ec481d8c65bc", "0x9ff6a4ec626c4cbc", "0x52131ecd506117f0", "0x9a7db1822213f9b9", "0x025ac827f2f88c7c",
|
||||
"0x4076da8ca02131e1", "0xbc9956bc70d53e0b", "0x077cf7d297357750", "0xb00c8db428fbccf2", "0x50f16eecd1fb65c4", "0x4daddb3cf455583d", "0x952b3cca95e87c92", "0xa7b7af6eac1a0222",
|
||||
"0xc58c0db5a99ced05", "0xa71a8ce697d65e94", "0xe29bab54459076d2", "0x5f613619a76c6400", "0xb43e9559e242a8d4", "0x4b5433e68aa1f302", "0x382b7f105840032c", "0xbf402649fb8e9618",
|
||||
"0xf45994bb29726a41", "0x8a11f358c24795cb", "0xc2c4f8902007527c", "0xe12f65396a832dcd", "0x307d3f495790aff0", "0x5fc00c5eb0c3e81e", "0xb46600e4685191ee", "0xce65db0e1d36f875"
|
||||
]}
|
||||
],
|
||||
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],
|
||||
|
|
@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
|
|||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -214,7 +214,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
|
|
@ -226,14 +226,14 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
{ uint s_ = r7 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
|
||||
r3 = r3 ^ ds[r7 & mask]; // 23 load
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
|
|
@ -242,19 +242,19 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint s_ = r5 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
|
||||
r3 = r3 ^ ds[r5 & mask]; // 44 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
|
|
@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
|
|
@ -44,7 +44,7 @@ __global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItem
|
|||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -52,7 +52,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -74,7 +74,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
|
||||
|
|
@ -86,14 +86,14 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = __umulhi(r2, r5); // 22 mulhi
|
||||
{ uint32_t s_ = r7 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
|
||||
r3 = r3 ^ ds[r7 & mask]; // 23 load
|
||||
r7 = __umulhi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
|
|
@ -102,19 +102,19 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint32_t s_ = r5 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
|
||||
r3 = r3 ^ ds[r5 & mask]; // 44 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
|
|
@ -129,7 +129,7 @@ __global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonc
|
|||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
{ uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
|
|
@ -177,7 +177,7 @@ __kernel void igneum_build(__global uint* ds, __global const uint* cache, uint n
|
|||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -185,7 +185,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -214,7 +214,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
|
|
@ -226,14 +226,14 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
{ uint s_ = r7 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
|
||||
r3 = r3 ^ ds[r7 & mask]; // 23 load
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
|
|
@ -242,19 +242,19 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint s_ = r5 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
|
||||
r3 = r3 ^ ds[r5 & mask]; // 44 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
|
|
@ -269,7 +269,7 @@ IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out
|
|||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
|
|
@ -295,7 +295,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -325,7 +325,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
|
|
@ -337,14 +337,14 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
{ uint s_ = r7 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
|
||||
r3 = r3 ^ ds[r7 & mask]; // 23 load
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
|
|
@ -353,19 +353,19 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint s_ = r5 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
|
||||
r3 = r3 ^ ds[r5 & mask]; // 44 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
|
|
@ -380,7 +380,7 @@ IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulon
|
|||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 2047u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
|
|
@ -20,7 +20,7 @@ __device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
|||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -28,7 +28,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -50,7 +50,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
|
||||
|
|
@ -62,14 +62,14 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = __umulhi(r2, r5); // 22 mulhi
|
||||
{ uint32_t s_ = r7 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
|
||||
r3 = r3 ^ ds[r7 & mask]; // 23 load
|
||||
r7 = __umulhi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
|
|
@ -78,19 +78,19 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint32_t s_ = r0 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint32_t s_ = r5 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
|
||||
r3 = r3 ^ ds[r5 & mask]; // 44 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
|
|
@ -105,7 +105,7 @@ __global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t ba
|
|||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint32_t s_ = r4 & 2047u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
{ uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
67
proto-cuda/packs-readwidth/scr4k128/program.h
Normal file
67
proto-cuda/packs-readwidth/scr4k128/program.h
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
|
||||
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
#include <cuda_runtime.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_SEED_STRING "igneum-genesis"
|
||||
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
|
||||
#define IGNEUM_GENERATOR 2
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0xe0c444d155cd78cfull
|
||||
#define IGNEUM_DAY_STRING "2026-10-03"
|
||||
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
|
||||
#define IGNEUM_DAY0 0x3067619fu
|
||||
#define IGNEUM_DAY1 0x3c269176u
|
||||
#define IGNEUM_DATASET_LOG2 28
|
||||
#define IGNEUM_MASK 0x0fffffffu
|
||||
#define IGNEUM_LANES 32
|
||||
#define IGNEUM_ITERATIONS 8
|
||||
#define IGNEUM_INSTR_COUNT 64
|
||||
#define IGNEUM_LOADS_PER_HASH 128
|
||||
#define IGNEUM_WIDE_LOADS_PER_HASH 0
|
||||
#define IGNEUM_OP_MIX "load=12 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 scratch=4 shfl=4 rotr=2 rotl=1"
|
||||
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
|
||||
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
|
||||
#define IGNEUM_LOAD_CLASS "scr4k128"
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 12, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 384
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// Variant 5: persistent warps, a 128 KiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
|
||||
#define IGNEUM_PERSISTENT_WARPS 1
|
||||
#define IGNEUM_SCRATCH_OPS 4 // scratch read-modify-writes per program (32 per hash)
|
||||
#define IGNEUM_SCRATCH_SLOTS 256u
|
||||
#define IGNEUM_SCRATCH_WORDS_PER_LANE 1024u
|
||||
#define IGNEUM_SCRATCH_BYTES_PER_WARP 131072u
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }
|
||||
#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u }
|
||||
#define IGNEUM_CACHE_LOG2_WORDS 26
|
||||
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
|
||||
#define IGNEUM_CACHE_SEGMENTS 65536u
|
||||
#define IGNEUM_ITEM_ROUNDS 8
|
||||
#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u }
|
||||
#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u }
|
||||
#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u }
|
||||
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt);
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#endif
|
||||
|
|
@ -2,7 +2,7 @@
|
|||
"format": "igneum-program-pack-3",
|
||||
"generator": 2,
|
||||
"attempt": 0,
|
||||
"program_id": "0x2f098ee568f386f5",
|
||||
"program_id": "0xe0c444d155cd78cf",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-genesis",
|
||||
|
|
@ -15,13 +15,14 @@
|
|||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"load_class": "scr4",
|
||||
"load_class": "scr4k128",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [12, 0, 0],
|
||||
"bytes_per_hash": 384,
|
||||
"scratch_ops_per_hash": 32,
|
||||
"scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
|
||||
"scratch_kib_per_warp": 128,
|
||||
"scratch": "variant 5 (measurement only): persistent warps; a 128 KiB scratch per warp of 256 16-byte slots per lane (lane-major); slot = src & 0xff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"op_mix": {"load": 12, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "scratch": 4, "shfl": 4, "rotr": 2, "rotl": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
126
proto-cuda/packs-readwidth/scr4k128/program.metal
Normal file
126
proto-cuda/packs-readwidth/scr4k128/program.metal
Normal file
|
|
@ -0,0 +1,126 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
device uint* scratch [[buffer(3)]],
|
||||
constant uint& groups [[buffer(4)]],
|
||||
constant uint& salt [[buffer(5)]],
|
||||
uint tid [[thread_position_in_grid]],
|
||||
uint nthreads [[threads_per_grid]]) {
|
||||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
|
||||
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
|
||||
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
|
||||
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
|
||||
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
|
||||
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
|
||||
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
|
||||
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
|
||||
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
|
||||
r4 = r0 * r6 + r4; // 3
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 4
|
||||
r4 = r4 ^ dataset[r1 & MASK]; // 5
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
|
||||
r7 = r7 ^ r5; // 8
|
||||
r3 = r3 | r4; // 9
|
||||
r1 = r1 | r2; // 10
|
||||
r4 = r4 ^ dataset[r3 & MASK]; // 11
|
||||
r6 = r6 | r2; // 12
|
||||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
|
||||
r5 = r5 ^ r7; // 21
|
||||
r2 = mulhi(r2, r5); // 22
|
||||
r3 = r3 ^ dataset[r7 & MASK]; // 23
|
||||
r7 = mulhi(r7, r3); // 24
|
||||
r5 = r5 | r4; // 25
|
||||
r4 = r5 * r2 + r4; // 26
|
||||
r5 = r5 * r1; // 27
|
||||
r6 = mulhi(r6, r7); // 28
|
||||
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
|
||||
r6 = rotr_var(r6, r7); // 30
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
r4 = r4 ^ dataset[r0 & MASK]; // 37
|
||||
r1 = r3 * r5 + r1; // 38
|
||||
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
|
||||
r2 = rotr_var(r2, r5); // 40
|
||||
r3 = r3 * r2; // 41
|
||||
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
|
||||
r3 = r3 ^ r4; // 43
|
||||
r3 = r3 ^ dataset[r5 & MASK]; // 44
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
|
||||
r7 = r7 ^ r1; // 46
|
||||
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
|
||||
r7 = mulhi(r7, r5); // 48
|
||||
r0 = r0 ^ dataset[r2 & MASK]; // 49
|
||||
r2 = r2 - r6; // 50
|
||||
r7 = r7 - r5; // 51
|
||||
r2 = r2 ^ r3; // 52
|
||||
r7 = r7 - r0; // 53
|
||||
r3 = r5 * r0 + r3; // 54
|
||||
r7 = r7 ^ r5; // 55
|
||||
r2 = r2 ^ dataset[r7 & MASK]; // 56
|
||||
r5 = r5 - r6; // 57
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 58
|
||||
{ uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
r4 = r4 - r6; // 60
|
||||
r1 = r1 * r2; // 61
|
||||
r3 = r6 * r0 + r3; // 62
|
||||
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
128
proto-cuda/packs-readwidth/scr4k128/program_bound.metal
Normal file
128
proto-cuda/packs-readwidth/scr4k128/program_bound.metal
Normal file
|
|
@ -0,0 +1,128 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
|
||||
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
constant uint* initw [[buffer(3)]],
|
||||
device uint* scratch [[buffer(4)]],
|
||||
constant uint& groups [[buffer(5)]],
|
||||
constant uint& salt [[buffer(6)]],
|
||||
uint tid [[thread_position_in_grid]],
|
||||
uint nthreads [[threads_per_grid]]) {
|
||||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
|
||||
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
|
||||
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
|
||||
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
|
||||
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
|
||||
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
|
||||
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
|
||||
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
|
||||
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
|
||||
r4 = r0 * r6 + r4; // 3
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 4
|
||||
r4 = r4 ^ dataset[r1 & MASK]; // 5
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
|
||||
r7 = r7 ^ r5; // 8
|
||||
r3 = r3 | r4; // 9
|
||||
r1 = r1 | r2; // 10
|
||||
r4 = r4 ^ dataset[r3 & MASK]; // 11
|
||||
r6 = r6 | r2; // 12
|
||||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
|
||||
r5 = r5 ^ r7; // 21
|
||||
r2 = mulhi(r2, r5); // 22
|
||||
r3 = r3 ^ dataset[r7 & MASK]; // 23
|
||||
r7 = mulhi(r7, r3); // 24
|
||||
r5 = r5 | r4; // 25
|
||||
r4 = r5 * r2 + r4; // 26
|
||||
r5 = r5 * r1; // 27
|
||||
r6 = mulhi(r6, r7); // 28
|
||||
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
|
||||
r6 = rotr_var(r6, r7); // 30
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
r4 = r4 ^ dataset[r0 & MASK]; // 37
|
||||
r1 = r3 * r5 + r1; // 38
|
||||
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
|
||||
r2 = rotr_var(r2, r5); // 40
|
||||
r3 = r3 * r2; // 41
|
||||
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
|
||||
r3 = r3 ^ r4; // 43
|
||||
r3 = r3 ^ dataset[r5 & MASK]; // 44
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
|
||||
r7 = r7 ^ r1; // 46
|
||||
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
|
||||
r7 = mulhi(r7, r5); // 48
|
||||
r0 = r0 ^ dataset[r2 & MASK]; // 49
|
||||
r2 = r2 - r6; // 50
|
||||
r7 = r7 - r5; // 51
|
||||
r2 = r2 ^ r3; // 52
|
||||
r7 = r7 - r0; // 53
|
||||
r3 = r5 * r0 + r3; // 54
|
||||
r7 = r7 ^ r5; // 55
|
||||
r2 = r2 ^ dataset[r7 & MASK]; // 56
|
||||
r5 = r5 - r6; // 57
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 58
|
||||
{ uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
r4 = r4 - r6; // 60
|
||||
r1 = r1 * r2; // 61
|
||||
r3 = r6 * r0 + r3; // 62
|
||||
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
|
|
@ -11,22 +11,22 @@
|
|||
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
|
||||
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
||||
{ // base nonce 0
|
||||
0x2273e2732203e32aull, 0xa513d354bd107990ull, 0xe005b7515054c85full, 0x18a61b37b30cd1fbull, 0xa21d5b98e8d07e9cull, 0x5c24171a391d5ed0ull, 0x0f29e7583e1794b8ull, 0x7eca8a374d1a4f70ull,
|
||||
0x260ca011cd9ea10cull, 0xce1050798fce3d43ull, 0x9560d939dca19041ull, 0x8480a8440b80ecc3ull, 0xfaa99aac459b739eull, 0x7f083e72458e08abull, 0x78d876842f68672bull, 0x3b9bcf6275d3575cull,
|
||||
0x0256af61bdbf11b3ull, 0xefc6771cae646cbdull, 0xbc44f1c9f9f54d87ull, 0x6caedd783487eb7dull, 0x001b31fcb4fe0d4dull, 0x947a7ba1057e25b6ull, 0xb9e5a0204d68c22aull, 0x50489bed25d42661ull,
|
||||
0xb1019bff6d1057cdull, 0xd1442990562ce940ull, 0xcd986a47f98801dbull, 0x9c8796b6df23300full, 0xbd53ad05d2c877a9ull, 0xc95e863774a15b0aull, 0x132d8a91fb2fa67aull, 0x53dd38e8eadc8a24ull
|
||||
0xd48ade5043a1a440ull, 0x0236be04051c86abull, 0x4a368a9d6ce0d6eaull, 0x58518f225df7cd40ull, 0xa82cf7b3417df902ull, 0x4f2722d25e5aa401ull, 0xf94296c41a24bdb4ull, 0x4fca15537288398bull,
|
||||
0x9ba276d89178f795ull, 0x7205ccbf3772e4e5ull, 0x4149e0fbf1c35bb9ull, 0x91d484090f0b09e0ull, 0xb95010c64c2d27dbull, 0x2b9fc3c52f733771ull, 0x1f2f6dd44045e10cull, 0xb78170533d4b16a6ull,
|
||||
0x679056d6c824110full, 0x0f81ba5b1476c314ull, 0xeb3d83ce6ccd9cd2ull, 0x370575fe0b0e7119ull, 0x7b4ae5a0f119315full, 0x12ceac820c840cfcull, 0xd190a33169dd6d61ull, 0xf7f90f768fc7ec83ull,
|
||||
0x90f11a170fce1e73ull, 0xca168b60b8a41d61ull, 0xab9da4f155dc3d4cull, 0x884d93725ea8fc2full, 0x202bf861848ba637ull, 0x7d508d345e67589eull, 0x8dfcbd9365f96ddaull, 0x5e8315d79af30be3ull
|
||||
},
|
||||
{ // base nonce 4096
|
||||
0x78c93312a03fb0eeull, 0x184fea638ec9b5fbull, 0x5687e8dcc4301dbfull, 0xed02c94f23681dfcull, 0x326d70162241ff6dull, 0x452017eb4ed2dfcfull, 0xc10b0e016f1e28c9ull, 0x691ce0cecf2a99baull,
|
||||
0x6c9506f34e0e63ceull, 0x447a98c2b7fdfa40ull, 0x07486b0e4055b2c9ull, 0x41781460bd47fd5cull, 0x01db316e35198291ull, 0xccd7e727f139a880ull, 0xdd7bd9efd16bf21cull, 0x8285d37966656366ull,
|
||||
0x383deade15fe0ecbull, 0x5fd64f5873c8e324ull, 0xad584cb6839c5e1dull, 0xbb842707fb5e9460ull, 0x4e8bc8f87978fcbdull, 0x18eb56f4a1fae881ull, 0x4c3b731a6b0c47a1ull, 0xda52cf9d69b252ebull,
|
||||
0xb5ff19b2b3eeb13eull, 0xe2595cea2afe42ddull, 0x3ff108424c9e6e38ull, 0x3a8a9e1995f359caull, 0x6a6b1da662cf2126ull, 0x54e684c127bb181full, 0x2018caa81f1a7d50ull, 0x33e94d2c92d9d148ull
|
||||
0x0925cd0a405af0f8ull, 0x9eeae6619738a6a0ull, 0x69d83343d36fa441ull, 0xef6e0dda67f22db7ull, 0xf5a5baddc99fc6e8ull, 0x048243c3a33d6313ull, 0xa2ed984433185d72ull, 0x5ce6444f5132231eull,
|
||||
0xf9c92e489ee1479bull, 0x13df97d418a1bb1dull, 0x54d8a4aa14eb5bf2ull, 0xc93ebe91c3aa1860ull, 0x12e1ca6f27af870bull, 0xa37cd8c938ec675bull, 0x0085b0d9144040d0ull, 0x389ce17c45d36ec5ull,
|
||||
0xc82d6694437e2f54ull, 0xf5a7b7357bc34eb6ull, 0x5e8e9c4cdaedb41dull, 0x18bff888b957603aull, 0x661b918790cf28f7ull, 0xf1411538bea6c80full, 0x0b3f28dd15dd2a0full, 0x076cd4e3230c6857ull,
|
||||
0x2cf5028d8fe8a18full, 0x270327a0333a8520ull, 0x53461f279f163486ull, 0x834a13f157378136ull, 0x5f8609aa7fbda1aeull, 0xd57be373dec55a73ull, 0xdf6b4c6134656905ull, 0x607a5b7e2a33765eull
|
||||
},
|
||||
{ // base nonce 1000000
|
||||
0x04a41389bf3dfd3dull, 0xc509164def9207dfull, 0x4a8ffdbdf46e429dull, 0xff13bf0dc1b39aebull, 0xb852acc8e24133d7ull, 0x4bdd991ae56252acull, 0xa7739e74b3a054e9ull, 0xb4e36218d4b45fdcull,
|
||||
0x8bbd323155f5edc5ull, 0xb7b56a90659e7fd2ull, 0xdff7c495b7027480ull, 0xffa8adb5c0302b06ull, 0xe97d7967d89a5672ull, 0x0d0c2d4e6493926eull, 0xe9a5cda333cf2043ull, 0xdc95256d0986e5d8ull,
|
||||
0xd0dc211b811d6843ull, 0x68dfa3d0fb9a569bull, 0xa9e0028dfd9178c0ull, 0x4a36ca1fc40b20a9ull, 0xe7c765c5a735294bull, 0xf08954b015cb2628ull, 0xc69ee66ecf2740c5ull, 0xe3d01e899e46b089ull,
|
||||
0xc3558c74159c8603ull, 0x4c7aeb196bd01b04ull, 0x13c17119385f1910ull, 0xda7fca98e0989b8aull, 0x6f95baf340817945ull, 0x1af52756fd3afcabull, 0xb8eefc370bbe7e4bull, 0x94a55055b48bd4dbull
|
||||
0x16b8e21167f3437cull, 0xfd28ad2d0f75a03cull, 0xf8e70bdb604cfff7ull, 0xeba037043c5ece4bull, 0xa2cb7d31f4d25317ull, 0xc3e8b85a50bdab1dull, 0xd7bd65a4353eab2eull, 0x23540281fae8cce3ull,
|
||||
0x37f4deac1a67cc5full, 0xd482f81bec2535a5ull, 0xc18f3f46f812b870ull, 0x582514aab0cf566dull, 0xb3b1a7424a758ac6ull, 0x83bbf70ed4151fa4ull, 0x72e2fed205f44a00ull, 0x2f81d15c1a8e17feull,
|
||||
0xbe7875e7927ca851ull, 0x1a72a20d292cc17bull, 0x859dd2c75675a04bull, 0xe12711e4d81b1e04ull, 0xfafefd6afdda6b35ull, 0x30ebb4d12f5cf4e3ull, 0x6d4aae24eee724a2ull, 0x317d83e64bf5e9c6ull,
|
||||
0x2beba8ecb8b0b26eull, 0x5cf2eacb7a58bd99ull, 0x2d56441aac88a037ull, 0x01708cc58adfeb95ull, 0xb3bc095b95418a2full, 0xf804c273322638e1ull, 0x9d89d48818056f24ull, 0xe07cbfd9aa54ccf1ull
|
||||
}
|
||||
};
|
||||
|
||||
|
|
@ -8,22 +8,22 @@
|
|||
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
|
||||
"warps": [
|
||||
{"base_nonce": 0, "expected": [
|
||||
"0x66cffcc97c46e625", "0xbc9019f8df50fbfd", "0x65629c90dde6016e", "0xd647a41effa03d3b", "0x86da3b6bbd751b99", "0x6ccf4240a0fb2d19", "0xeb39a1e06f17378c", "0x2ea6b349b289fb10",
|
||||
"0x6067211e6c220500", "0x6e6095dfedd1360f", "0xbd1190d8b50e1b48", "0x216dc72a0c08d5b5", "0x5be1f8c080836b0c", "0x2a32932a5953ed73", "0xcc2a3d68be83c802", "0xe46daca15338278f",
|
||||
"0xb2de43b96761e459", "0x9004acd06588cbea", "0x6a9a3543cf93004f", "0xff956d859cb6e408", "0x4397ec6e3c5fb045", "0x521dea569cd481d5", "0x89832b34108759f0", "0xf66e393836ffe4ea",
|
||||
"0xb4e39af6c40ea2f4", "0x3adc22085dd8d648", "0x27efe270958bbfbb", "0x6c80be0e8dca60d8", "0xa0afbc6a60260d59", "0x5d9a257fb9189537", "0xeb837aeef55dc3ed", "0xd381174dc14f8951"
|
||||
"0xd48ade5043a1a440", "0x0236be04051c86ab", "0x4a368a9d6ce0d6ea", "0x58518f225df7cd40", "0xa82cf7b3417df902", "0x4f2722d25e5aa401", "0xf94296c41a24bdb4", "0x4fca15537288398b",
|
||||
"0x9ba276d89178f795", "0x7205ccbf3772e4e5", "0x4149e0fbf1c35bb9", "0x91d484090f0b09e0", "0xb95010c64c2d27db", "0x2b9fc3c52f733771", "0x1f2f6dd44045e10c", "0xb78170533d4b16a6",
|
||||
"0x679056d6c824110f", "0x0f81ba5b1476c314", "0xeb3d83ce6ccd9cd2", "0x370575fe0b0e7119", "0x7b4ae5a0f119315f", "0x12ceac820c840cfc", "0xd190a33169dd6d61", "0xf7f90f768fc7ec83",
|
||||
"0x90f11a170fce1e73", "0xca168b60b8a41d61", "0xab9da4f155dc3d4c", "0x884d93725ea8fc2f", "0x202bf861848ba637", "0x7d508d345e67589e", "0x8dfcbd9365f96dda", "0x5e8315d79af30be3"
|
||||
]},
|
||||
{"base_nonce": 4096, "expected": [
|
||||
"0xfb1f61aaeaeaef94", "0x184c2160963d8b57", "0x42a053c628625778", "0xeaad0e41c4770812", "0x1d6d389ceb462ce1", "0x4a639827672bdbd4", "0x857e42aa5a42dd6f", "0xb4ef399e5339979c",
|
||||
"0x497a29225b099233", "0x71d8b42862d81954", "0x0af995663313bf04", "0xf436fd126619d7a1", "0x199e4e3333cff269", "0x64077952f3775768", "0x51af1d126c5e8388", "0xffbaf44fe6b15cfd",
|
||||
"0xfc8fed86ecae34a7", "0x4cb548616f7a7d6b", "0xc21d938c8b5bef35", "0x34789cbdd7088f71", "0xacb099a2c207d891", "0xfe1902d162374413", "0x26f7831c28f4020b", "0xdf5192952b4af6b0",
|
||||
"0xecab61fe88dbaff4", "0x941c491f7fdb86e5", "0x2b1900c53f746e77", "0x8c40507b1caffeb2", "0x7532a1ec2b9169ef", "0x1cf399b0c8bfb520", "0xdf003d2bb8a2cc0c", "0x4da853307fc977a9"
|
||||
"0x0925cd0a405af0f8", "0x9eeae6619738a6a0", "0x69d83343d36fa441", "0xef6e0dda67f22db7", "0xf5a5baddc99fc6e8", "0x048243c3a33d6313", "0xa2ed984433185d72", "0x5ce6444f5132231e",
|
||||
"0xf9c92e489ee1479b", "0x13df97d418a1bb1d", "0x54d8a4aa14eb5bf2", "0xc93ebe91c3aa1860", "0x12e1ca6f27af870b", "0xa37cd8c938ec675b", "0x0085b0d9144040d0", "0x389ce17c45d36ec5",
|
||||
"0xc82d6694437e2f54", "0xf5a7b7357bc34eb6", "0x5e8e9c4cdaedb41d", "0x18bff888b957603a", "0x661b918790cf28f7", "0xf1411538bea6c80f", "0x0b3f28dd15dd2a0f", "0x076cd4e3230c6857",
|
||||
"0x2cf5028d8fe8a18f", "0x270327a0333a8520", "0x53461f279f163486", "0x834a13f157378136", "0x5f8609aa7fbda1ae", "0xd57be373dec55a73", "0xdf6b4c6134656905", "0x607a5b7e2a33765e"
|
||||
]},
|
||||
{"base_nonce": 1000000, "expected": [
|
||||
"0x3d094bd04694b96f", "0xaecbd76cecd1a20a", "0xbcb86febe56b17fe", "0x98082b557ba97517", "0xbb5f94108888564b", "0xea3284877a30fc87", "0xc608fa4d5a8bb2ad", "0x946c721c511e0729",
|
||||
"0x46c6eba292083aed", "0x936cb97231eb6795", "0xb1413c434c712cbb", "0xedfd554d3948c1bd", "0xa8a20cbef2faccd5", "0x5fe39d756cadbcad", "0x208b2627380791fe", "0xf52f9374ce480218",
|
||||
"0xb9db7cd8814eb29e", "0xf32ed2192b5a8719", "0x4f1b06a054940aef", "0x406df498e4365eb5", "0x1982075caad345ef", "0x590f725623dbbbbd", "0xa26d9192dedfefa5", "0x36219ec00da18980",
|
||||
"0x5361d0dcb0f8b1a3", "0x35bdefa2fbb5ffc3", "0xba4c2a4e473a9c80", "0x107d9d3030f8b9d3", "0xa8bb094266d6b987", "0x86164fdfbb1426e8", "0xa6e8cb895021cbbd", "0xfe12809e9d99a243"
|
||||
"0x16b8e21167f3437c", "0xfd28ad2d0f75a03c", "0xf8e70bdb604cfff7", "0xeba037043c5ece4b", "0xa2cb7d31f4d25317", "0xc3e8b85a50bdab1d", "0xd7bd65a4353eab2e", "0x23540281fae8cce3",
|
||||
"0x37f4deac1a67cc5f", "0xd482f81bec2535a5", "0xc18f3f46f812b870", "0x582514aab0cf566d", "0xb3b1a7424a758ac6", "0x83bbf70ed4151fa4", "0x72e2fed205f44a00", "0x2f81d15c1a8e17fe",
|
||||
"0xbe7875e7927ca851", "0x1a72a20d292cc17b", "0x859dd2c75675a04b", "0xe12711e4d81b1e04", "0xfafefd6afdda6b35", "0x30ebb4d12f5cf4e3", "0x6d4aae24eee724a2", "0x317d83e64bf5e9c6",
|
||||
"0x2beba8ecb8b0b26e", "0x5cf2eacb7a58bd99", "0x2d56441aac88a037", "0x01708cc58adfeb95", "0xb3bc095b95418a2f", "0xf804c273322638e1", "0x9d89d48818056f24", "0xe07cbfd9aa54ccf1"
|
||||
]}
|
||||
],
|
||||
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],
|
||||
291
proto-cuda/packs-readwidth/scr4k32/kernel.cl
Normal file
291
proto-cuda/packs-readwidth/scr4k32/kernel.cl
Normal file
|
|
@ -0,0 +1,291 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
|
||||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
r3 = r3 ^ ds[r7 & mask]; // 23 load
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = mul_hi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
r3 = r3 ^ ds[r5 & mask]; // 44 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = mul_hi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
177
proto-cuda/packs-readwidth/scr4k32/kernel.cu
Normal file
177
proto-cuda/packs-readwidth/scr4k32/kernel.cu
Normal file
|
|
@ -0,0 +1,177 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
|
||||
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
#include "memhard.h"
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
|
||||
uint32_t x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
|
||||
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
|
||||
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (t < nItems) {
|
||||
uint32_t s[16];
|
||||
mh_item(cache, t, s);
|
||||
uint32_t* d = ds + (size_t)t * 16u;
|
||||
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, uint32_t* scratch, uint32_t groups, uint32_t salt) {
|
||||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
uint32_t tag = salt + g_;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = __umulhi(r2, r5); // 22 mulhi
|
||||
r3 = r3 ^ ds[r7 & mask]; // 23 load
|
||||
r7 = __umulhi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = __umulhi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
r3 = r3 ^ ds[r5 & mask]; // 44 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = __umulhi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
}
|
||||
|
||||
// Host-side launch wrappers. Declared in program.h, called from host.cu.
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
if (nSegments == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nSegments + block - 1u) / block;
|
||||
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
if (nItems == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nItems + block - 1u) / block;
|
||||
igneum_build<<<grid, block>>>(ds, cache, nItems);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
|
||||
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, scratch, nonces / 32u, salt);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
393
proto-cuda/packs-readwidth/scr4k32/kernel_bound.cl
Normal file
393
proto-cuda/packs-readwidth/scr4k32/kernel_bound.cl
Normal file
|
|
@ -0,0 +1,393 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
|
||||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
r3 = r3 ^ ds[r7 & mask]; // 23 load
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = mul_hi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
r3 = r3 ^ ds[r5 & mask]; // 44 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = mul_hi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
|
||||
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
|
||||
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global uint* scratch, uint groups, uint salt) {
|
||||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
|
||||
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
|
||||
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
|
||||
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
|
||||
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
|
||||
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
|
||||
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
|
||||
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
r3 = r3 ^ ds[r7 & mask]; // 23 load
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = mul_hi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
r3 = r3 ^ ds[r5 & mask]; // 44 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = mul_hi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
136
proto-cuda/packs-readwidth/scr4k32/kernel_bound.cu
Normal file
136
proto-cuda/packs-readwidth/scr4k32/kernel_bound.cu
Normal file
|
|
@ -0,0 +1,136 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
|
||||
// Host declarations (also in program_bound.h if present):
|
||||
// struct IgneumInitWords { uint32_t w[8]; };
|
||||
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
|
||||
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
|
||||
struct IgneumInitWords { uint32_t w[8]; };
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, uint32_t* scratch, uint32_t groups, uint32_t salt) {
|
||||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
uint32_t tag = salt + g_;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
|
||||
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
|
||||
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
|
||||
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
|
||||
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
|
||||
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
|
||||
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
|
||||
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
r7 = r7 ^ ds[r2 & mask]; // 4 load
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = __umulhi(r2, r5); // 22 mulhi
|
||||
r3 = r3 ^ ds[r7 & mask]; // 23 load
|
||||
r7 = __umulhi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = __umulhi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
r4 = r4 ^ ds[r0 & mask]; // 37 load
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
r3 = r3 ^ ds[r5 & mask]; // 44 load
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = __umulhi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
}
|
||||
|
||||
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
|
||||
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
|
||||
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash_bound<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, iw, scratch, nonces / 32u, salt);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
108
proto-cuda/packs-readwidth/scr4k32/memhard.h
Normal file
108
proto-cuda/packs-readwidth/scr4k32/memhard.h
Normal file
|
|
@ -0,0 +1,108 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
|
||||
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
|
||||
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#if defined(__CUDACC__)
|
||||
#define IGNEUM_HD __host__ __device__ __forceinline__
|
||||
#elif defined(_MSC_VER) && !defined(__cplusplus)
|
||||
#define IGNEUM_HD static __inline
|
||||
#else
|
||||
#define IGNEUM_HD static inline
|
||||
#endif
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint32_t r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
|
||||
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
|
||||
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint32_t r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
106
proto-cuda/packs-readwidth/scr4k32/memhard.metal
Normal file
106
proto-cuda/packs-readwidth/scr4k32/memhard.metal
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
inline void mh_cache_segment(device uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
inline void mh_mixer(thread uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// One thread per segment (2^16 threads).
|
||||
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
|
||||
mh_cache_segment(cache, gid);
|
||||
}
|
||||
// One thread per 64-byte item (dataset words / 16 threads).
|
||||
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint s[16];
|
||||
mh_item(cache, gid, s);
|
||||
device uint* d = dataset + gid * 16u;
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
|
|
@ -15,7 +15,7 @@
|
|||
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
|
||||
#define IGNEUM_GENERATOR 2
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0x2f098ee568f386f5ull
|
||||
#define IGNEUM_PROGRAM_ID 0xe0c4a4d155ce1befull
|
||||
#define IGNEUM_DAY_STRING "2026-10-03"
|
||||
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
|
||||
#define IGNEUM_DAY0 0x3067619fu
|
||||
|
|
@ -30,20 +30,20 @@
|
|||
#define IGNEUM_OP_MIX "load=12 add=9 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 scratch=4 shfl=4 rotr=2 rotl=1"
|
||||
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
|
||||
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
|
||||
#define IGNEUM_LOAD_CLASS "scr4"
|
||||
#define IGNEUM_LOAD_CLASS "scr4k32"
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 12, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 384
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
|
||||
#define IGNEUM_PERSISTENT_WARPS 1
|
||||
#define IGNEUM_SCRATCH_OPS 4 // scratch read-modify-writes per program (32 per hash)
|
||||
#define IGNEUM_SCRATCH_SLOTS 2048u
|
||||
#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u
|
||||
#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u
|
||||
#define IGNEUM_SCRATCH_SLOTS 64u
|
||||
#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u
|
||||
#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
130
proto-cuda/packs-readwidth/scr4k32/program.json
Normal file
130
proto-cuda/packs-readwidth/scr4k32/program.json
Normal file
|
|
@ -0,0 +1,130 @@
|
|||
{
|
||||
"format": "igneum-program-pack-3",
|
||||
"generator": 2,
|
||||
"attempt": 0,
|
||||
"program_id": "0xe0c4a4d155ce1bef",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-genesis",
|
||||
"seed_bytes": "69676e65756d2d67656e65736973",
|
||||
"seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"],
|
||||
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
|
||||
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
|
||||
"lanes": 32,
|
||||
"registers": 8,
|
||||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"load_class": "scr4k32",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [12, 0, 0],
|
||||
"bytes_per_hash": 384,
|
||||
"scratch_ops_per_hash": 32,
|
||||
"scratch_kib_per_warp": 32,
|
||||
"scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"op_mix": {"load": 12, "add": 9, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "scratch": 4, "shfl": 4, "rotr": 2, "rotl": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
|
||||
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
|
||||
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
|
||||
"op_semantics": {
|
||||
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
|
||||
"sub": "dst = dst - src",
|
||||
"mul": "dst = dst * src (low 32)",
|
||||
"mulhi": "dst = high 32 bits of dst * src",
|
||||
"xor": "dst = dst ^ src",
|
||||
"or": "dst = dst | src",
|
||||
"rotl": "dst = rotl(dst, rot), rot in 1..31",
|
||||
"rotr": "dst = rotr(dst, src & 31)",
|
||||
"mad": "dst = src * src2 + dst",
|
||||
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
|
||||
"load": "dst = dst ^ dataset[src & dataset.mask]",
|
||||
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
|
||||
},
|
||||
"dataset": {
|
||||
"log2_words": 28,
|
||||
"bytes": 1073741824,
|
||||
"mask": "0x0fffffff",
|
||||
"day": "2026-10-03",
|
||||
"day_bytes": "6461792f323032362d31302d3033",
|
||||
"day_words_from": "seed_words_from_bytes(day_bytes)",
|
||||
"d0": "0x3067619f",
|
||||
"d1": "0x3c269176",
|
||||
"mode": "memory-hard",
|
||||
"spec": "proto-metal/MEMHARD.md",
|
||||
"key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"],
|
||||
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
|
||||
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
|
||||
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
|
||||
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s",
|
||||
"word": "dataset[w] = item(w >> 4)[w & 15]"
|
||||
},
|
||||
"instructions": [
|
||||
{"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1},
|
||||
{"i": 1, "op": "add", "dst": 1, "src": 7, "src2": 2, "imm": "0x42da7657", "imm2": "0xc3bd2355", "rot": 25, "bit": 4, "mask": 16, "width": 1},
|
||||
{"i": 2, "op": "add", "dst": 2, "src": 3, "src2": 0, "imm": "0x61f0b51c", "imm2": "0x2735a174", "rot": 4, "bit": 26, "mask": 2, "width": 1},
|
||||
{"i": 3, "op": "mad", "dst": 4, "src": 0, "src2": 6, "imm": "0x679648a8", "imm2": "0x3044ba32", "rot": 31, "bit": 31, "mask": 4, "width": 1},
|
||||
{"i": 4, "op": "load", "dst": 7, "src": 2, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1},
|
||||
{"i": 5, "op": "load", "dst": 4, "src": 1, "src2": 2, "imm": "0x987c017a", "imm2": "0xf4d60559", "rot": 2, "bit": 0, "mask": 4, "width": 1},
|
||||
{"i": 6, "op": "shfl", "dst": 6, "src": 3, "src2": 7, "imm": "0x6ea7b2df", "imm2": "0x9fce5071", "rot": 7, "bit": 15, "mask": 4, "width": 1},
|
||||
{"i": 7, "op": "shfl", "dst": 1, "src": 5, "src2": 1, "imm": "0x26a2ecde", "imm2": "0xfec6ad22", "rot": 15, "bit": 11, "mask": 8, "width": 1},
|
||||
{"i": 8, "op": "xor", "dst": 7, "src": 5, "src2": 2, "imm": "0xbe4b445c", "imm2": "0x17a5a9c7", "rot": 8, "bit": 8, "mask": 1, "width": 1},
|
||||
{"i": 9, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1},
|
||||
{"i": 10, "op": "or", "dst": 1, "src": 2, "src2": 3, "imm": "0x4e7dc10d", "imm2": "0x196d165c", "rot": 14, "bit": 27, "mask": 16, "width": 1},
|
||||
{"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 1, "imm": "0xc5c3b55d", "imm2": "0xec061424", "rot": 26, "bit": 27, "mask": 8, "width": 1},
|
||||
{"i": 12, "op": "or", "dst": 6, "src": 2, "src2": 3, "imm": "0x306542fe", "imm2": "0x1bb1b429", "rot": 31, "bit": 0, "mask": 2, "width": 1},
|
||||
{"i": 13, "op": "mul", "dst": 2, "src": 5, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1},
|
||||
{"i": 14, "op": "load", "dst": 1, "src": 2, "src2": 5, "imm": "0x028b4d37", "imm2": "0x7bbd78ea", "rot": 15, "bit": 2, "mask": 8, "width": 1},
|
||||
{"i": 15, "op": "rotl", "dst": 7, "src": 6, "src2": 6, "imm": "0x5c88a1a7", "imm2": "0x5c628769", "rot": 1, "bit": 3, "mask": 8, "width": 1},
|
||||
{"i": 16, "op": "scratch", "dst": 3, "src": 6, "src2": 7, "imm": "0xbac2ae81", "imm2": "0xcbbc7bdb", "rot": 18, "bit": 8, "mask": 8, "width": 1},
|
||||
{"i": 17, "op": "load", "dst": 7, "src": 4, "src2": 2, "imm": "0xe8ab93e9", "imm2": "0xa00de107", "rot": 2, "bit": 1, "mask": 16, "width": 1},
|
||||
{"i": 18, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1},
|
||||
{"i": 19, "op": "mad", "dst": 4, "src": 0, "src2": 2, "imm": "0x5fba7bc2", "imm2": "0xdf099cfb", "rot": 4, "bit": 15, "mask": 16, "width": 1},
|
||||
{"i": 20, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0x0a3056de", "imm2": "0x7f0c25c3", "rot": 27, "bit": 13, "mask": 8, "width": 1},
|
||||
{"i": 21, "op": "xor", "dst": 5, "src": 7, "src2": 4, "imm": "0xbd066e1d", "imm2": "0x6d3ddc5a", "rot": 2, "bit": 29, "mask": 1, "width": 1},
|
||||
{"i": 22, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1},
|
||||
{"i": 23, "op": "load", "dst": 3, "src": 7, "src2": 2, "imm": "0xc7fcfc8f", "imm2": "0x8528b94f", "rot": 17, "bit": 13, "mask": 4, "width": 1},
|
||||
{"i": 24, "op": "mulhi", "dst": 7, "src": 3, "src2": 5, "imm": "0xd91641e8", "imm2": "0xaf77faf2", "rot": 22, "bit": 21, "mask": 1, "width": 1},
|
||||
{"i": 25, "op": "or", "dst": 5, "src": 4, "src2": 0, "imm": "0x84c03868", "imm2": "0xf6c691b7", "rot": 29, "bit": 14, "mask": 8, "width": 1},
|
||||
{"i": 26, "op": "mad", "dst": 4, "src": 5, "src2": 2, "imm": "0x3bb2b6ba", "imm2": "0x49d95fd5", "rot": 1, "bit": 5, "mask": 8, "width": 1},
|
||||
{"i": 27, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1},
|
||||
{"i": 28, "op": "mulhi", "dst": 6, "src": 7, "src2": 6, "imm": "0xd69c4715", "imm2": "0xe0ebc4ce", "rot": 29, "bit": 2, "mask": 8, "width": 1},
|
||||
{"i": 29, "op": "add", "dst": 6, "src": 1, "src2": 2, "imm": "0x3b2d2124", "imm2": "0x187a9128", "rot": 1, "bit": 9, "mask": 16, "width": 1},
|
||||
{"i": 30, "op": "rotr", "dst": 6, "src": 7, "src2": 0, "imm": "0x5c64a589", "imm2": "0x61c9a38d", "rot": 17, "bit": 21, "mask": 16, "width": 1},
|
||||
{"i": 31, "op": "load", "dst": 3, "src": 1, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1},
|
||||
{"i": 32, "op": "scratch", "dst": 1, "src": 0, "src2": 7, "imm": "0xcc7972c4", "imm2": "0xad098d15", "rot": 30, "bit": 21, "mask": 8, "width": 1},
|
||||
{"i": 33, "op": "add", "dst": 0, "src": 4, "src2": 4, "imm": "0x2c35699f", "imm2": "0x351dde38", "rot": 21, "bit": 18, "mask": 4, "width": 1},
|
||||
{"i": 34, "op": "scratch", "dst": 0, "src": 2, "src2": 3, "imm": "0xfae8902b", "imm2": "0x5cd8306f", "rot": 5, "bit": 28, "mask": 16, "width": 1},
|
||||
{"i": 35, "op": "mul", "dst": 0, "src": 3, "src2": 1, "imm": "0x4fa3f3db", "imm2": "0xdbf37e75", "rot": 7, "bit": 18, "mask": 4, "width": 1},
|
||||
{"i": 36, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1},
|
||||
{"i": 37, "op": "load", "dst": 4, "src": 0, "src2": 0, "imm": "0x04cc1d55", "imm2": "0x35c52d04", "rot": 11, "bit": 14, "mask": 2, "width": 1},
|
||||
{"i": 38, "op": "mad", "dst": 1, "src": 3, "src2": 5, "imm": "0x3958f280", "imm2": "0x8713c7e1", "rot": 5, "bit": 23, "mask": 16, "width": 1},
|
||||
{"i": 39, "op": "add", "dst": 0, "src": 3, "src2": 3, "imm": "0xa907b90b", "imm2": "0x1b053acf", "rot": 30, "bit": 25, "mask": 16, "width": 1},
|
||||
{"i": 40, "op": "rotr", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1},
|
||||
{"i": 41, "op": "mul", "dst": 3, "src": 2, "src2": 4, "imm": "0x49087d74", "imm2": "0x6348b489", "rot": 17, "bit": 9, "mask": 16, "width": 1},
|
||||
{"i": 42, "op": "add", "dst": 1, "src": 5, "src2": 1, "imm": "0xa32e000c", "imm2": "0x6058c2e3", "rot": 25, "bit": 20, "mask": 8, "width": 1},
|
||||
{"i": 43, "op": "xor", "dst": 3, "src": 4, "src2": 2, "imm": "0x3dad0eb6", "imm2": "0xb97578cb", "rot": 3, "bit": 27, "mask": 1, "width": 1},
|
||||
{"i": 44, "op": "load", "dst": 3, "src": 5, "src2": 7, "imm": "0x374aec92", "imm2": "0x626f11df", "rot": 20, "bit": 18, "mask": 8, "width": 1},
|
||||
{"i": 45, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1},
|
||||
{"i": 46, "op": "xor", "dst": 7, "src": 1, "src2": 0, "imm": "0xef6ac348", "imm2": "0x963bb7e6", "rot": 26, "bit": 3, "mask": 8, "width": 1},
|
||||
{"i": 47, "op": "add", "dst": 0, "src": 3, "src2": 0, "imm": "0x838b5065", "imm2": "0x36360066", "rot": 3, "bit": 31, "mask": 4, "width": 1},
|
||||
{"i": 48, "op": "mulhi", "dst": 7, "src": 5, "src2": 0, "imm": "0x8458f7ac", "imm2": "0xc1c15026", "rot": 27, "bit": 15, "mask": 8, "width": 1},
|
||||
{"i": 49, "op": "load", "dst": 0, "src": 2, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1},
|
||||
{"i": 50, "op": "sub", "dst": 2, "src": 6, "src2": 0, "imm": "0x2baec8c9", "imm2": "0x4390f156", "rot": 3, "bit": 12, "mask": 8, "width": 1},
|
||||
{"i": 51, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x19234061", "imm2": "0xe84dfade", "rot": 4, "bit": 19, "mask": 1, "width": 1},
|
||||
{"i": 52, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0xdc2cd71e", "imm2": "0x1b5d334b", "rot": 9, "bit": 8, "mask": 8, "width": 1},
|
||||
{"i": 53, "op": "sub", "dst": 7, "src": 0, "src2": 4, "imm": "0x605c31ec", "imm2": "0x9923ff88", "rot": 28, "bit": 25, "mask": 4, "width": 1},
|
||||
{"i": 54, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1},
|
||||
{"i": 55, "op": "xor", "dst": 7, "src": 5, "src2": 5, "imm": "0xad7493e7", "imm2": "0x3e400372", "rot": 13, "bit": 8, "mask": 1, "width": 1},
|
||||
{"i": 56, "op": "load", "dst": 2, "src": 7, "src2": 1, "imm": "0x87e933c9", "imm2": "0x8c854c1b", "rot": 17, "bit": 3, "mask": 8, "width": 1},
|
||||
{"i": 57, "op": "sub", "dst": 5, "src": 6, "src2": 5, "imm": "0x11be3bc9", "imm2": "0xbbaa8e24", "rot": 6, "bit": 5, "mask": 16, "width": 1},
|
||||
{"i": 58, "op": "load", "dst": 1, "src": 3, "src2": 2, "imm": "0xa732351a", "imm2": "0xc01349cd", "rot": 14, "bit": 17, "mask": 16, "width": 1},
|
||||
{"i": 59, "op": "scratch", "dst": 1, "src": 4, "src2": 0, "imm": "0xb20547b2", "imm2": "0xc94655de", "rot": 27, "bit": 30, "mask": 1, "width": 1},
|
||||
{"i": 60, "op": "sub", "dst": 4, "src": 6, "src2": 7, "imm": "0x67cf904c", "imm2": "0x6873b216", "rot": 27, "bit": 7, "mask": 16, "width": 1},
|
||||
{"i": 61, "op": "mul", "dst": 1, "src": 2, "src2": 7, "imm": "0x93ab0bf4", "imm2": "0x96158375", "rot": 14, "bit": 0, "mask": 16, "width": 1},
|
||||
{"i": 62, "op": "mad", "dst": 3, "src": 6, "src2": 0, "imm": "0x41a443a3", "imm2": "0xe69d7919", "rot": 9, "bit": 0, "mask": 16, "width": 1},
|
||||
{"i": 63, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x2fe0e98b", "imm2": "0xc88e2942", "rot": 5, "bit": 16, "mask": 16, "width": 1}
|
||||
]
|
||||
}
|
||||
126
proto-cuda/packs-readwidth/scr4k32/program.metal
Normal file
126
proto-cuda/packs-readwidth/scr4k32/program.metal
Normal file
|
|
@ -0,0 +1,126 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
device uint* scratch [[buffer(3)]],
|
||||
constant uint& groups [[buffer(4)]],
|
||||
constant uint& salt [[buffer(5)]],
|
||||
uint tid [[thread_position_in_grid]],
|
||||
uint nthreads [[threads_per_grid]]) {
|
||||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
|
||||
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
|
||||
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
|
||||
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
|
||||
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
|
||||
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
|
||||
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
|
||||
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
|
||||
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
|
||||
r4 = r0 * r6 + r4; // 3
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 4
|
||||
r4 = r4 ^ dataset[r1 & MASK]; // 5
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
|
||||
r7 = r7 ^ r5; // 8
|
||||
r3 = r3 | r4; // 9
|
||||
r1 = r1 | r2; // 10
|
||||
r4 = r4 ^ dataset[r3 & MASK]; // 11
|
||||
r6 = r6 | r2; // 12
|
||||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
|
||||
r5 = r5 ^ r7; // 21
|
||||
r2 = mulhi(r2, r5); // 22
|
||||
r3 = r3 ^ dataset[r7 & MASK]; // 23
|
||||
r7 = mulhi(r7, r3); // 24
|
||||
r5 = r5 | r4; // 25
|
||||
r4 = r5 * r2 + r4; // 26
|
||||
r5 = r5 * r1; // 27
|
||||
r6 = mulhi(r6, r7); // 28
|
||||
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
|
||||
r6 = rotr_var(r6, r7); // 30
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
r4 = r4 ^ dataset[r0 & MASK]; // 37
|
||||
r1 = r3 * r5 + r1; // 38
|
||||
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
|
||||
r2 = rotr_var(r2, r5); // 40
|
||||
r3 = r3 * r2; // 41
|
||||
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
|
||||
r3 = r3 ^ r4; // 43
|
||||
r3 = r3 ^ dataset[r5 & MASK]; // 44
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
|
||||
r7 = r7 ^ r1; // 46
|
||||
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
|
||||
r7 = mulhi(r7, r5); // 48
|
||||
r0 = r0 ^ dataset[r2 & MASK]; // 49
|
||||
r2 = r2 - r6; // 50
|
||||
r7 = r7 - r5; // 51
|
||||
r2 = r2 ^ r3; // 52
|
||||
r7 = r7 - r0; // 53
|
||||
r3 = r5 * r0 + r3; // 54
|
||||
r7 = r7 ^ r5; // 55
|
||||
r2 = r2 ^ dataset[r7 & MASK]; // 56
|
||||
r5 = r5 - r6; // 57
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 58
|
||||
{ uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
r4 = r4 - r6; // 60
|
||||
r1 = r1 * r2; // 61
|
||||
r3 = r6 * r0 + r3; // 62
|
||||
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
128
proto-cuda/packs-readwidth/scr4k32/program_bound.metal
Normal file
128
proto-cuda/packs-readwidth/scr4k32/program_bound.metal
Normal file
|
|
@ -0,0 +1,128 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
|
||||
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
constant uint* initw [[buffer(3)]],
|
||||
device uint* scratch [[buffer(4)]],
|
||||
constant uint& groups [[buffer(5)]],
|
||||
constant uint& salt [[buffer(6)]],
|
||||
uint tid [[thread_position_in_grid]],
|
||||
uint nthreads [[threads_per_grid]]) {
|
||||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
|
||||
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
|
||||
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
|
||||
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
|
||||
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
|
||||
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
|
||||
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
|
||||
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
|
||||
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
|
||||
r4 = r0 * r6 + r4; // 3
|
||||
r7 = r7 ^ dataset[r2 & MASK]; // 4
|
||||
r4 = r4 ^ dataset[r1 & MASK]; // 5
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
|
||||
r7 = r7 ^ r5; // 8
|
||||
r3 = r3 | r4; // 9
|
||||
r1 = r1 | r2; // 10
|
||||
r4 = r4 ^ dataset[r3 & MASK]; // 11
|
||||
r6 = r6 | r2; // 12
|
||||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
|
||||
r5 = r5 ^ r7; // 21
|
||||
r2 = mulhi(r2, r5); // 22
|
||||
r3 = r3 ^ dataset[r7 & MASK]; // 23
|
||||
r7 = mulhi(r7, r3); // 24
|
||||
r5 = r5 | r4; // 25
|
||||
r4 = r5 * r2 + r4; // 26
|
||||
r5 = r5 * r1; // 27
|
||||
r6 = mulhi(r6, r7); // 28
|
||||
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
|
||||
r6 = rotr_var(r6, r7); // 30
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
r4 = r4 ^ dataset[r0 & MASK]; // 37
|
||||
r1 = r3 * r5 + r1; // 38
|
||||
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
|
||||
r2 = rotr_var(r2, r5); // 40
|
||||
r3 = r3 * r2; // 41
|
||||
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
|
||||
r3 = r3 ^ r4; // 43
|
||||
r3 = r3 ^ dataset[r5 & MASK]; // 44
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
|
||||
r7 = r7 ^ r1; // 46
|
||||
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
|
||||
r7 = mulhi(r7, r5); // 48
|
||||
r0 = r0 ^ dataset[r2 & MASK]; // 49
|
||||
r2 = r2 - r6; // 50
|
||||
r7 = r7 - r5; // 51
|
||||
r2 = r2 ^ r3; // 52
|
||||
r7 = r7 - r0; // 53
|
||||
r3 = r5 * r0 + r3; // 54
|
||||
r7 = r7 ^ r5; // 55
|
||||
r2 = r2 ^ dataset[r7 & MASK]; // 56
|
||||
r5 = r5 - r6; // 57
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 58
|
||||
{ uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
r4 = r4 - r6; // 60
|
||||
r1 = r1 * r2; // 61
|
||||
r3 = r6 * r0 + r3; // 62
|
||||
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
57
proto-cuda/packs-readwidth/scr4k32/vectors.h
Normal file
57
proto-cuda/packs-readwidth/scr4k32/vectors.h
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_VEC_WARPS 3
|
||||
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
|
||||
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
||||
{ // base nonce 0
|
||||
0x62cab4be0ed880e1ull, 0x90842c849268cf52ull, 0x53d4d591f4a12749ull, 0x020429c3d1279eddull, 0x7086876f3a9183fbull, 0xc1f954e8065b6d02ull, 0x7550abd24b6ed9edull, 0x7dcc57f17b255fc9ull,
|
||||
0x01f16667b6ea326dull, 0x05907ad2b28423a5ull, 0x27e8aad8889ea702ull, 0xe93e61d2ff565fabull, 0x21aa77db95a8790full, 0xdc361b8011f7a395ull, 0xbabf590a4cf6dd58ull, 0x7a9b7eb46cf2b88dull,
|
||||
0xeb43f11d39490205ull, 0x3c2e3b2ed2b0c6fbull, 0x2bc2293577914bc8ull, 0x66cc059287de6cc0ull, 0x9a2d3a0f30169b20ull, 0xa5756e5027ec459full, 0x734a7fb8546a08b5ull, 0xbc59502ef67511d7ull,
|
||||
0x869d40616fa13209ull, 0x7e2713e23d7c3e06ull, 0x640f81856492a8d3ull, 0x1f7ce7b42962bc41ull, 0xc474fdd339993867ull, 0xb08d199c21d08397ull, 0x6b021b07d5dd9fdfull, 0x574adb548a4f3be8ull
|
||||
},
|
||||
{ // base nonce 4096
|
||||
0x2956e7703c1553fcull, 0xe06e9c5dd64f0cffull, 0x41d967b788797c3eull, 0xc9beec7571ab7808ull, 0x0d0d99d51ac72942ull, 0xac60f79ae1bb46d6ull, 0xd9b9a509bc33d145ull, 0x512d444977d257f5ull,
|
||||
0x0e9759a10d773b28ull, 0xb270e841265b2b3dull, 0x90b97771870e53dcull, 0xf8b0bacb1ead0c1bull, 0x32165ca85108736cull, 0x5a907f0cb371d6d1ull, 0x89d4ccf9b6323847ull, 0x495e23339db371dbull,
|
||||
0xa50d559aa7911894ull, 0xbe561e2c2e64f0ffull, 0x862b141ae3b898eaull, 0x69b52c3068f0544aull, 0x2769f3b4051f9e80ull, 0xb679a28f140a5ccaull, 0x037d194732dce935ull, 0xec9f1e85406dbee9ull,
|
||||
0x65a0d3f10857795dull, 0x5da0b4908b5cda66ull, 0x1cbdf4f47dad39e4ull, 0x5472317d40d55545ull, 0x24ec2fb5eeff7691ull, 0x4c56104e2454b9beull, 0x8b896d9e85dbf491ull, 0xe80882d5975e09ecull
|
||||
},
|
||||
{ // base nonce 1000000
|
||||
0xa417c0e0494f5f0dull, 0x44cfa8bbf55cb40bull, 0xee534b970664a105ull, 0x2814b1857db92d67ull, 0xd358f7e47b35ff5full, 0xa08faa47e58221c3ull, 0xbb76559b9a4a447bull, 0xd438a3fd1fa2976eull,
|
||||
0xa08d0e2c88abe900ull, 0x58b3c3ab097c416dull, 0x705de177cf28ccdcull, 0x263f35e27d8cf3aaull, 0xa2c304ccfeb9ae9bull, 0x470ba6ea4e8f661aull, 0xa59e5f33cd8613d9ull, 0xdb887848353dc91cull,
|
||||
0xd948d1c36b6a98e1ull, 0xd78806062c54882aull, 0x0744e029194938aaull, 0x42b613ec3d9074c6ull, 0x88c75044753d1496ull, 0x940239d09cbd80e1ull, 0x4534dd536f53adf5ull, 0x6f44b9bd4d6564deull,
|
||||
0xb6d8142c422857d5ull, 0x3b61f0c20b8fb04bull, 0x17eaf3a88b49c9fdull, 0xec536015d760eb1eull, 0xf37e9ad57045cc05ull, 0x588b808bc29ab6caull, 0x7cfa7ae3c6e483c2ull, 0x71e3cc45c07b530cull
|
||||
}
|
||||
};
|
||||
|
||||
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
|
||||
static const uint32_t IGNEUM_DS_HEAD[16] = {
|
||||
0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu,
|
||||
0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
|
||||
static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u;
|
||||
// 64 sampled dataset words (index, value) computed on the Mac.
|
||||
#define IGNEUM_DS_SAMPLES 64
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
|
||||
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
|
||||
0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u
|
||||
};
|
||||
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
|
||||
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
|
||||
0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u,
|
||||
0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u
|
||||
};
|
||||
static const uint32_t IGNEUM_CACHE_LAST[16] = {
|
||||
0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du,
|
||||
0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu
|
||||
};
|
||||
static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull;
|
||||
36
proto-cuda/packs-readwidth/scr4k32/vectors.json
Normal file
36
proto-cuda/packs-readwidth/scr4k32/vectors.json
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
{
|
||||
"seed": "igneum-genesis",
|
||||
"day": "2026-10-03",
|
||||
"dataset_mode": "memory-hard",
|
||||
"dataset_log2_words": 28,
|
||||
"mask": "0x0fffffff",
|
||||
"lanes": 32,
|
||||
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
|
||||
"warps": [
|
||||
{"base_nonce": 0, "expected": [
|
||||
"0x62cab4be0ed880e1", "0x90842c849268cf52", "0x53d4d591f4a12749", "0x020429c3d1279edd", "0x7086876f3a9183fb", "0xc1f954e8065b6d02", "0x7550abd24b6ed9ed", "0x7dcc57f17b255fc9",
|
||||
"0x01f16667b6ea326d", "0x05907ad2b28423a5", "0x27e8aad8889ea702", "0xe93e61d2ff565fab", "0x21aa77db95a8790f", "0xdc361b8011f7a395", "0xbabf590a4cf6dd58", "0x7a9b7eb46cf2b88d",
|
||||
"0xeb43f11d39490205", "0x3c2e3b2ed2b0c6fb", "0x2bc2293577914bc8", "0x66cc059287de6cc0", "0x9a2d3a0f30169b20", "0xa5756e5027ec459f", "0x734a7fb8546a08b5", "0xbc59502ef67511d7",
|
||||
"0x869d40616fa13209", "0x7e2713e23d7c3e06", "0x640f81856492a8d3", "0x1f7ce7b42962bc41", "0xc474fdd339993867", "0xb08d199c21d08397", "0x6b021b07d5dd9fdf", "0x574adb548a4f3be8"
|
||||
]},
|
||||
{"base_nonce": 4096, "expected": [
|
||||
"0x2956e7703c1553fc", "0xe06e9c5dd64f0cff", "0x41d967b788797c3e", "0xc9beec7571ab7808", "0x0d0d99d51ac72942", "0xac60f79ae1bb46d6", "0xd9b9a509bc33d145", "0x512d444977d257f5",
|
||||
"0x0e9759a10d773b28", "0xb270e841265b2b3d", "0x90b97771870e53dc", "0xf8b0bacb1ead0c1b", "0x32165ca85108736c", "0x5a907f0cb371d6d1", "0x89d4ccf9b6323847", "0x495e23339db371db",
|
||||
"0xa50d559aa7911894", "0xbe561e2c2e64f0ff", "0x862b141ae3b898ea", "0x69b52c3068f0544a", "0x2769f3b4051f9e80", "0xb679a28f140a5cca", "0x037d194732dce935", "0xec9f1e85406dbee9",
|
||||
"0x65a0d3f10857795d", "0x5da0b4908b5cda66", "0x1cbdf4f47dad39e4", "0x5472317d40d55545", "0x24ec2fb5eeff7691", "0x4c56104e2454b9be", "0x8b896d9e85dbf491", "0xe80882d5975e09ec"
|
||||
]},
|
||||
{"base_nonce": 1000000, "expected": [
|
||||
"0xa417c0e0494f5f0d", "0x44cfa8bbf55cb40b", "0xee534b970664a105", "0x2814b1857db92d67", "0xd358f7e47b35ff5f", "0xa08faa47e58221c3", "0xbb76559b9a4a447b", "0xd438a3fd1fa2976e",
|
||||
"0xa08d0e2c88abe900", "0x58b3c3ab097c416d", "0x705de177cf28ccdc", "0x263f35e27d8cf3aa", "0xa2c304ccfeb9ae9b", "0x470ba6ea4e8f661a", "0xa59e5f33cd8613d9", "0xdb887848353dc91c",
|
||||
"0xd948d1c36b6a98e1", "0xd78806062c54882a", "0x0744e029194938aa", "0x42b613ec3d9074c6", "0x88c75044753d1496", "0x940239d09cbd80e1", "0x4534dd536f53adf5", "0x6f44b9bd4d6564de",
|
||||
"0xb6d8142c422857d5", "0x3b61f0c20b8fb04b", "0x17eaf3a88b49c9fd", "0xec536015d760eb1e", "0xf37e9ad57045cc05", "0x588b808bc29ab6ca", "0x7cfa7ae3c6e483c2", "0x71e3cc45c07b530c"
|
||||
]}
|
||||
],
|
||||
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],
|
||||
"dataset_last_index": 268435455,
|
||||
"dataset_last": "0xa33ada72",
|
||||
"dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}],
|
||||
"cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"],
|
||||
"cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"],
|
||||
"cache_fnv1a64": "0x48c4f5bf24166b2e"
|
||||
}
|
||||
291
proto-cuda/packs-readwidth/scr8k128/kernel.cl
Normal file
291
proto-cuda/packs-readwidth/scr8k128/kernel.cl
Normal file
|
|
@ -0,0 +1,291 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
|
||||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
{ uint s_ = r7 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = mul_hi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint s_ = r5 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = mul_hi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
177
proto-cuda/packs-readwidth/scr8k128/kernel.cu
Normal file
177
proto-cuda/packs-readwidth/scr8k128/kernel.cu
Normal file
|
|
@ -0,0 +1,177 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
|
||||
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
#include "memhard.h"
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
|
||||
uint32_t x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
|
||||
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
|
||||
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (t < nItems) {
|
||||
uint32_t s[16];
|
||||
mh_item(cache, t, s);
|
||||
uint32_t* d = ds + (size_t)t * 16u;
|
||||
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, uint32_t* scratch, uint32_t groups, uint32_t salt) {
|
||||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
uint32_t tag = salt + g_;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = __umulhi(r2, r5); // 22 mulhi
|
||||
{ uint32_t s_ = r7 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
|
||||
r7 = __umulhi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = __umulhi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint32_t s_ = r5 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = __umulhi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
}
|
||||
|
||||
// Host-side launch wrappers. Declared in program.h, called from host.cu.
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
if (nSegments == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nSegments + block - 1u) / block;
|
||||
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
if (nItems == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nItems + block - 1u) / block;
|
||||
igneum_build<<<grid, block>>>(ds, cache, nItems);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
|
||||
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, scratch, nonces / 32u, salt);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
393
proto-cuda/packs-readwidth/scr8k128/kernel_bound.cl
Normal file
393
proto-cuda/packs-readwidth/scr8k128/kernel_bound.cl
Normal file
|
|
@ -0,0 +1,393 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
|
||||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
{ uint s_ = r7 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = mul_hi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint s_ = r5 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = mul_hi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
|
||||
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
|
||||
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global uint* scratch, uint groups, uint salt) {
|
||||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
|
||||
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
|
||||
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
|
||||
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
|
||||
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
|
||||
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
|
||||
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
|
||||
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
{ uint s_ = r7 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = mul_hi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint s_ = r5 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = mul_hi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 255u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
136
proto-cuda/packs-readwidth/scr8k128/kernel_bound.cu
Normal file
136
proto-cuda/packs-readwidth/scr8k128/kernel_bound.cu
Normal file
|
|
@ -0,0 +1,136 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
|
||||
// Host declarations (also in program_bound.h if present):
|
||||
// struct IgneumInitWords { uint32_t w[8]; };
|
||||
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
|
||||
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
|
||||
struct IgneumInitWords { uint32_t w[8]; };
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, uint32_t* scratch, uint32_t groups, uint32_t salt) {
|
||||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
uint32_t tag = salt + g_;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
|
||||
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
|
||||
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
|
||||
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
|
||||
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
|
||||
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
|
||||
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
|
||||
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = __umulhi(r2, r5); // 22 mulhi
|
||||
{ uint32_t s_ = r7 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
|
||||
r7 = __umulhi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = __umulhi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint32_t s_ = r0 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint32_t s_ = r5 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = __umulhi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint32_t s_ = r4 & 255u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
}
|
||||
|
||||
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
|
||||
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
|
||||
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash_bound<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, iw, scratch, nonces / 32u, salt);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
108
proto-cuda/packs-readwidth/scr8k128/memhard.h
Normal file
108
proto-cuda/packs-readwidth/scr8k128/memhard.h
Normal file
|
|
@ -0,0 +1,108 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
|
||||
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
|
||||
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#if defined(__CUDACC__)
|
||||
#define IGNEUM_HD __host__ __device__ __forceinline__
|
||||
#elif defined(_MSC_VER) && !defined(__cplusplus)
|
||||
#define IGNEUM_HD static __inline
|
||||
#else
|
||||
#define IGNEUM_HD static inline
|
||||
#endif
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint32_t r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
|
||||
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
|
||||
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint32_t r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
106
proto-cuda/packs-readwidth/scr8k128/memhard.metal
Normal file
106
proto-cuda/packs-readwidth/scr8k128/memhard.metal
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
inline void mh_cache_segment(device uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
inline void mh_mixer(thread uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// One thread per segment (2^16 threads).
|
||||
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
|
||||
mh_cache_segment(cache, gid);
|
||||
}
|
||||
// One thread per 64-byte item (dataset words / 16 threads).
|
||||
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint s[16];
|
||||
mh_item(cache, gid, s);
|
||||
device uint* d = dataset + gid * 16u;
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
67
proto-cuda/packs-readwidth/scr8k128/program.h
Normal file
67
proto-cuda/packs-readwidth/scr8k128/program.h
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.
|
||||
// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
#include <cuda_runtime.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_SEED_STRING "igneum-genesis"
|
||||
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
|
||||
#define IGNEUM_GENERATOR 2
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0xe0d1dcd155d90573ull
|
||||
#define IGNEUM_DAY_STRING "2026-10-03"
|
||||
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
|
||||
#define IGNEUM_DAY0 0x3067619fu
|
||||
#define IGNEUM_DAY1 0x3c269176u
|
||||
#define IGNEUM_DATASET_LOG2 28
|
||||
#define IGNEUM_MASK 0x0fffffffu
|
||||
#define IGNEUM_LANES 32
|
||||
#define IGNEUM_ITERATIONS 8
|
||||
#define IGNEUM_INSTR_COUNT 64
|
||||
#define IGNEUM_LOADS_PER_HASH 128
|
||||
#define IGNEUM_WIDE_LOADS_PER_HASH 0
|
||||
#define IGNEUM_OP_MIX "add=9 load=8 scratch=8 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 rotl=1"
|
||||
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
|
||||
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
|
||||
#define IGNEUM_LOAD_CLASS "scr8k128"
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 8, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 256
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// Variant 5: persistent warps, a 128 KiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
|
||||
#define IGNEUM_PERSISTENT_WARPS 1
|
||||
#define IGNEUM_SCRATCH_OPS 8 // scratch read-modify-writes per program (64 per hash)
|
||||
#define IGNEUM_SCRATCH_SLOTS 256u
|
||||
#define IGNEUM_SCRATCH_WORDS_PER_LANE 1024u
|
||||
#define IGNEUM_SCRATCH_BYTES_PER_WARP 131072u
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
#define IGNEUM_SEEDW_INIT { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u }
|
||||
#define IGNEUM_KEY_INIT { 0x3067619fu, 0x3c269176u, 0x84a03b03u, 0xf8c63294u, 0xff977c5bu, 0xe60def3eu, 0x63630141u, 0xb8fbcb58u }
|
||||
#define IGNEUM_CACHE_LOG2_WORDS 26
|
||||
#define IGNEUM_CACHE_SEGMENT_LOG2_LINES 6
|
||||
#define IGNEUM_CACHE_SEGMENTS 65536u
|
||||
#define IGNEUM_ITEM_ROUNDS 8
|
||||
#define IGNEUM_MIX_ROT_INIT { 20u, 20u, 19u, 4u, 26u, 3u, 3u, 27u }
|
||||
#define IGNEUM_MIX_MUL_INIT { 0x42146205u, 0x52cbe0fbu, 0x7ecf4a03u, 0x6728907fu, 0xd81d9751u, 0x132952c3u, 0xf60de277u, 0x05358035u, 0xbaf6499du, 0xe4db9667u, 0x3e98f45du, 0xd0004eddu, 0x2691630du, 0x9beb3bcfu, 0xab310379u, 0x99cfb423u }
|
||||
#define IGNEUM_MIX_RC_INIT { 0xbab68293u, 0xcc162340u, 0x6ce151ccu, 0xe62b8997u, 0xc9c80297u, 0xf74a1654u, 0x3d704af5u, 0x3cf522b7u, 0x2b9cac04u, 0xa880ac10u, 0x13e5dd1du, 0x6fc3e233u, 0x2d83eeacu, 0x9006e8bfu, 0x2c4b5362u, 0x31b49ee2u }
|
||||
|
||||
#ifndef IGNEUM_NO_CUDA
|
||||
// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt);
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#endif
|
||||
|
|
@ -2,7 +2,7 @@
|
|||
"format": "igneum-program-pack-3",
|
||||
"generator": 2,
|
||||
"attempt": 0,
|
||||
"program_id": "0x2f0992e568f38dc1",
|
||||
"program_id": "0xe0d1dcd155d90573",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-genesis",
|
||||
|
|
@ -15,13 +15,14 @@
|
|||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"load_class": "scr8",
|
||||
"load_class": "scr8k128",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [8, 0, 0],
|
||||
"bytes_per_hash": 256,
|
||||
"scratch_ops_per_hash": 64,
|
||||
"scratch": "variant 5 (measurement only): persistent warps; a 1 MiB scratch per warp of 2048 16-byte slots per lane (lane-major); slot = src & 0x7ff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
|
||||
"scratch_kib_per_warp": 128,
|
||||
"scratch": "variant 5 (measurement only): persistent warps; a 128 KiB scratch per warp of 256 16-byte slots per lane (lane-major); slot = src & 0xff; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"op_mix": {"add": 9, "load": 8, "scratch": 8, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "rotl": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
|
|
@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
|
|||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -36,7 +36,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -58,7 +58,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
|
||||
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
|
||||
r4 = r0 * r6 + r4; // 3
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
|
||||
r4 = r4 ^ dataset[r1 & MASK]; // 5
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
|
||||
|
|
@ -70,14 +70,14 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
|
||||
r5 = r5 ^ r7; // 21
|
||||
r2 = mulhi(r2, r5); // 22
|
||||
{ uint s_ = r7 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
|
||||
{ uint s_ = r7 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
|
||||
r7 = mulhi(r7, r3); // 24
|
||||
r5 = r5 | r4; // 25
|
||||
r4 = r5 * r2 + r4; // 26
|
||||
|
|
@ -86,19 +86,19 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
|
||||
r6 = rotr_var(r6, r7); // 30
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
|
||||
r1 = r3 * r5 + r1; // 38
|
||||
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
|
||||
r2 = rotr_var(r2, r5); // 40
|
||||
r3 = r3 * r2; // 41
|
||||
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
|
||||
r3 = r3 ^ r4; // 43
|
||||
{ uint s_ = r5 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
|
||||
{ uint s_ = r5 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
|
||||
r7 = r7 ^ r1; // 46
|
||||
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
|
||||
|
|
@ -113,7 +113,7 @@ kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
|||
r2 = r2 ^ dataset[r7 & MASK]; // 56
|
||||
r5 = r5 - r6; // 57
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 58
|
||||
{ uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
{ uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
r4 = r4 - r6; // 60
|
||||
r1 = r1 * r2; // 61
|
||||
r3 = r6 * r0 + r3; // 62
|
||||
|
|
@ -21,7 +21,7 @@ inline uint ds_elem(uint i, uint d0, uint d1) {
|
|||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 1 MiB scratch per warp, 2048 slots of
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 128 KiB scratch per warp, 256 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
|
|
@ -38,7 +38,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 8192u;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 1024u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
|
|
@ -60,7 +60,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
|
||||
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
|
||||
r4 = r0 * r6 + r4; // 3
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
|
||||
r4 = r4 ^ dataset[r1 & MASK]; // 5
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
|
||||
|
|
@ -72,14 +72,14 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
{ uint s_ = r6 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
|
||||
r5 = r5 ^ r7; // 21
|
||||
r2 = mulhi(r2, r5); // 22
|
||||
{ uint s_ = r7 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
|
||||
{ uint s_ = r7 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
|
||||
r7 = mulhi(r7, r3); // 24
|
||||
r5 = r5 | r4; // 25
|
||||
r4 = r5 * r2 + r4; // 26
|
||||
|
|
@ -88,19 +88,19 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
|
||||
r6 = rotr_var(r6, r7); // 30
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
{ uint s_ = r2 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
{ uint s_ = r0 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
|
||||
{ uint s_ = r0 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
|
||||
r1 = r3 * r5 + r1; // 38
|
||||
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
|
||||
r2 = rotr_var(r2, r5); // 40
|
||||
r3 = r3 * r2; // 41
|
||||
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
|
||||
r3 = r3 ^ r4; // 43
|
||||
{ uint s_ = r5 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
|
||||
{ uint s_ = r5 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
|
||||
r7 = r7 ^ r1; // 46
|
||||
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
|
||||
|
|
@ -115,7 +115,7 @@ kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
|||
r2 = r2 ^ dataset[r7 & MASK]; // 56
|
||||
r5 = r5 - r6; // 57
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 58
|
||||
{ uint s_ = r4 & 2047u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
{ uint s_ = r4 & 255u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
r4 = r4 - r6; // 60
|
||||
r1 = r1 * r2; // 61
|
||||
r3 = r6 * r0 + r3; // 62
|
||||
57
proto-cuda/packs-readwidth/scr8k128/vectors.h
Normal file
57
proto-cuda/packs-readwidth/scr8k128/vectors.h
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_VEC_WARPS 3
|
||||
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
|
||||
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
||||
{ // base nonce 0
|
||||
0xe814771d17db365cull, 0xc4d4a4ae8b6033caull, 0xa4ea884e5e961196ull, 0x5a4f2e353f13502bull, 0xb5bfae78c5048e4cull, 0x75ed7ff2f8ccb9c1ull, 0xc9d37d26079b1916ull, 0xef3c29ccb9d46163ull,
|
||||
0xb4cae00d3e73ae8eull, 0x88bc44f26a90e913ull, 0xeb91c51e87da69f8ull, 0x2ee04eeac0ba97b2ull, 0x43e33706056bb735ull, 0x88acef8db41e6bbfull, 0x87295bf633750804ull, 0x4a8310fa3c5393f3ull,
|
||||
0x9c33366c7aa6a5a6ull, 0x79ba6d5674f78e5cull, 0x03168e07fb7ae416ull, 0xdd45a1b5f54270aeull, 0x0d0084aa74f95b51ull, 0xa04060d3f711930bull, 0xa080e7988516297full, 0xcefa3e2e8de2806full,
|
||||
0x95dc7a55e10c4010ull, 0x62e809b8cef37be6ull, 0x0366273048e795cfull, 0xc2b04c7deacb1dffull, 0xe83446f7db3a4686ull, 0xe6c7e6575c34651full, 0xee15067b419606d6ull, 0xfe67f3ae51faade1ull
|
||||
},
|
||||
{ // base nonce 4096
|
||||
0xdde6034f4b5824c9ull, 0x07b807430ab9effbull, 0xa581d141cb4bc74aull, 0x0a1b4129b3690618ull, 0xb4d10af1daaec58bull, 0xcf0b63a9aa6b8a96ull, 0x07c20bd30e3eb88cull, 0x32ebffcaafa5df9eull,
|
||||
0xe1b528deb263ddc3ull, 0xaa6d2e1e7f45c995ull, 0x6017aaa938e837cfull, 0x23445a9b8c8e5addull, 0x024ebd232a344f41ull, 0x67aebe3e79435f84ull, 0xa7d0b7522e88814aull, 0x1d4d9633b57ac637ull,
|
||||
0x77f500325912fcbdull, 0x9f4bc5d04fbb13b9ull, 0xd7e081a23934d582ull, 0x10992aa1c8a93afeull, 0x1596ba0b47520be7ull, 0x344ed3c63b5a75bcull, 0xdb75c50a7a39c7beull, 0x0e1ccc942ec1fad7ull,
|
||||
0x81c9d97c4d9605c6ull, 0x1e60918f98df7dc9ull, 0x3e60e5b90d94fe34ull, 0xec7163b65fc01cfdull, 0xb0786922940f66e3ull, 0xb1d049c3e24a38e2ull, 0x5a6a7ac9a0c8ed50ull, 0x7bc50eab43e83a01ull
|
||||
},
|
||||
{ // base nonce 1000000
|
||||
0x1ebe406e6227f5e9ull, 0xc9d07c89dd188990ull, 0x3346fbacbe00f719ull, 0x437f4d678259e06dull, 0xd664758bc7508b7cull, 0xa3428dfd2b480593ull, 0xdfbb1aca3c18aeb6ull, 0xe362f4d90b64ab9full,
|
||||
0xfaeac4630c51e291ull, 0x2b81c4de8eaa689aull, 0x661b54d4d8763782ull, 0x48839cd831ea402bull, 0xa98dbad17e5f3a49ull, 0x35161bdc5dc87db7ull, 0xcc503293dddde770ull, 0x8960f9fbf15a0e47ull,
|
||||
0x3d391f32613d80d5ull, 0xd3fb60a843c31905ull, 0x7f70cbe3a4be1f1eull, 0xb67d653a57d143c7ull, 0x08a9210687821d3dull, 0x54adfc465596fd4cull, 0x12cf1cdd36c931e4ull, 0x62fd59b6a002a406ull,
|
||||
0xfef506618454af46ull, 0x8ab9f4c86cfefbe0ull, 0xb54d7b600ee5cbdbull, 0x9400685a172c31ccull, 0x0cdc0ca4f87996dcull, 0x050be1f1631ab65full, 0x79884be8f2e9a1f2ull, 0x824d03dfe16bcb91ull
|
||||
}
|
||||
};
|
||||
|
||||
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
|
||||
static const uint32_t IGNEUM_DS_HEAD[16] = {
|
||||
0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu,
|
||||
0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
|
||||
static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u;
|
||||
// 64 sampled dataset words (index, value) computed on the Mac.
|
||||
#define IGNEUM_DS_SAMPLES 64
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
|
||||
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
|
||||
0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u
|
||||
};
|
||||
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
|
||||
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
|
||||
0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u,
|
||||
0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u
|
||||
};
|
||||
static const uint32_t IGNEUM_CACHE_LAST[16] = {
|
||||
0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du,
|
||||
0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu
|
||||
};
|
||||
static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull;
|
||||
36
proto-cuda/packs-readwidth/scr8k128/vectors.json
Normal file
36
proto-cuda/packs-readwidth/scr8k128/vectors.json
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
{
|
||||
"seed": "igneum-genesis",
|
||||
"day": "2026-10-03",
|
||||
"dataset_mode": "memory-hard",
|
||||
"dataset_log2_words": 28,
|
||||
"mask": "0x0fffffff",
|
||||
"lanes": 32,
|
||||
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
|
||||
"warps": [
|
||||
{"base_nonce": 0, "expected": [
|
||||
"0xe814771d17db365c", "0xc4d4a4ae8b6033ca", "0xa4ea884e5e961196", "0x5a4f2e353f13502b", "0xb5bfae78c5048e4c", "0x75ed7ff2f8ccb9c1", "0xc9d37d26079b1916", "0xef3c29ccb9d46163",
|
||||
"0xb4cae00d3e73ae8e", "0x88bc44f26a90e913", "0xeb91c51e87da69f8", "0x2ee04eeac0ba97b2", "0x43e33706056bb735", "0x88acef8db41e6bbf", "0x87295bf633750804", "0x4a8310fa3c5393f3",
|
||||
"0x9c33366c7aa6a5a6", "0x79ba6d5674f78e5c", "0x03168e07fb7ae416", "0xdd45a1b5f54270ae", "0x0d0084aa74f95b51", "0xa04060d3f711930b", "0xa080e7988516297f", "0xcefa3e2e8de2806f",
|
||||
"0x95dc7a55e10c4010", "0x62e809b8cef37be6", "0x0366273048e795cf", "0xc2b04c7deacb1dff", "0xe83446f7db3a4686", "0xe6c7e6575c34651f", "0xee15067b419606d6", "0xfe67f3ae51faade1"
|
||||
]},
|
||||
{"base_nonce": 4096, "expected": [
|
||||
"0xdde6034f4b5824c9", "0x07b807430ab9effb", "0xa581d141cb4bc74a", "0x0a1b4129b3690618", "0xb4d10af1daaec58b", "0xcf0b63a9aa6b8a96", "0x07c20bd30e3eb88c", "0x32ebffcaafa5df9e",
|
||||
"0xe1b528deb263ddc3", "0xaa6d2e1e7f45c995", "0x6017aaa938e837cf", "0x23445a9b8c8e5add", "0x024ebd232a344f41", "0x67aebe3e79435f84", "0xa7d0b7522e88814a", "0x1d4d9633b57ac637",
|
||||
"0x77f500325912fcbd", "0x9f4bc5d04fbb13b9", "0xd7e081a23934d582", "0x10992aa1c8a93afe", "0x1596ba0b47520be7", "0x344ed3c63b5a75bc", "0xdb75c50a7a39c7be", "0x0e1ccc942ec1fad7",
|
||||
"0x81c9d97c4d9605c6", "0x1e60918f98df7dc9", "0x3e60e5b90d94fe34", "0xec7163b65fc01cfd", "0xb0786922940f66e3", "0xb1d049c3e24a38e2", "0x5a6a7ac9a0c8ed50", "0x7bc50eab43e83a01"
|
||||
]},
|
||||
{"base_nonce": 1000000, "expected": [
|
||||
"0x1ebe406e6227f5e9", "0xc9d07c89dd188990", "0x3346fbacbe00f719", "0x437f4d678259e06d", "0xd664758bc7508b7c", "0xa3428dfd2b480593", "0xdfbb1aca3c18aeb6", "0xe362f4d90b64ab9f",
|
||||
"0xfaeac4630c51e291", "0x2b81c4de8eaa689a", "0x661b54d4d8763782", "0x48839cd831ea402b", "0xa98dbad17e5f3a49", "0x35161bdc5dc87db7", "0xcc503293dddde770", "0x8960f9fbf15a0e47",
|
||||
"0x3d391f32613d80d5", "0xd3fb60a843c31905", "0x7f70cbe3a4be1f1e", "0xb67d653a57d143c7", "0x08a9210687821d3d", "0x54adfc465596fd4c", "0x12cf1cdd36c931e4", "0x62fd59b6a002a406",
|
||||
"0xfef506618454af46", "0x8ab9f4c86cfefbe0", "0xb54d7b600ee5cbdb", "0x9400685a172c31cc", "0x0cdc0ca4f87996dc", "0x050be1f1631ab65f", "0x79884be8f2e9a1f2", "0x824d03dfe16bcb91"
|
||||
]}
|
||||
],
|
||||
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],
|
||||
"dataset_last_index": 268435455,
|
||||
"dataset_last": "0xa33ada72",
|
||||
"dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}],
|
||||
"cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"],
|
||||
"cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"],
|
||||
"cache_fnv1a64": "0x48c4f5bf24166b2e"
|
||||
}
|
||||
291
proto-cuda/packs-readwidth/scr8k32/kernel.cl
Normal file
291
proto-cuda/packs-readwidth/scr8k32/kernel.cl
Normal file
|
|
@ -0,0 +1,291 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
|
||||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
{ uint s_ = r7 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = mul_hi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint s_ = r5 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = mul_hi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
177
proto-cuda/packs-readwidth/scr8k32/kernel.cu
Normal file
177
proto-cuda/packs-readwidth/scr8k32/kernel.cu
Normal file
|
|
@ -0,0 +1,177 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).
|
||||
// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
#include "memhard.h"
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {
|
||||
uint32_t x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.
|
||||
// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.
|
||||
__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (t < nItems) {
|
||||
uint32_t s[16];
|
||||
mh_item(cache, t, s);
|
||||
uint32_t* d = ds + (size_t)t * 16u;
|
||||
for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every
|
||||
// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a
|
||||
// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, uint32_t* scratch, uint32_t groups, uint32_t salt) {
|
||||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
uint32_t tag = salt + g_;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint32_t x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint32_t x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint32_t x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint32_t x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint32_t x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint32_t x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint32_t x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = __umulhi(r2, r5); // 22 mulhi
|
||||
{ uint32_t s_ = r7 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
|
||||
r7 = __umulhi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = __umulhi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint32_t s_ = r5 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = __umulhi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
}
|
||||
|
||||
// Host-side launch wrappers. Declared in program.h, called from host.cu.
|
||||
cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {
|
||||
if (nSegments == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nSegments + block - 1u) / block;
|
||||
igneum_cache_fill<<<grid, block>>>(cache, nSegments);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {
|
||||
if (nItems == 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 256u;
|
||||
uint32_t grid = (nItems + block - 1u) / block;
|
||||
igneum_build<<<grid, block>>>(ds, cache, nItems);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
|
||||
cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
|
||||
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, scratch, nonces / 32u, salt);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
393
proto-cuda/packs-readwidth/scr8k32/kernel_bound.cl
Normal file
393
proto-cuda/packs-readwidth/scr8k32/kernel_bound.cl
Normal file
|
|
@ -0,0 +1,393 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).
|
||||
// Built from source at runtime by proto-opencl/host.c, which passes these defines:
|
||||
// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)
|
||||
// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)
|
||||
// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32
|
||||
// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition
|
||||
// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;
|
||||
// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.
|
||||
#ifndef IGNEUM_GROUP
|
||||
#define IGNEUM_GROUP 32
|
||||
#endif
|
||||
#ifndef IGNEUM_EXCHANGE
|
||||
#define IGNEUM_EXCHANGE 0
|
||||
#endif
|
||||
#ifdef __OPENCL_VERSION__
|
||||
#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))
|
||||
#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#ifdef cl_khr_subgroups
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroups : enable
|
||||
#endif
|
||||
#ifdef cl_khr_subgroup_shuffle
|
||||
#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable
|
||||
#endif
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#pragma OPENCL EXTENSION cl_intel_subgroups : enable
|
||||
#endif
|
||||
#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))
|
||||
#else
|
||||
// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.
|
||||
#include "emu_opencl.h"
|
||||
#endif
|
||||
|
||||
#if IGNEUM_EXCHANGE == 1
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#elif IGNEUM_EXCHANGE == 2
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))
|
||||
#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)
|
||||
#else
|
||||
// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per
|
||||
// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane
|
||||
// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's
|
||||
// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.
|
||||
#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }
|
||||
#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }
|
||||
#endif
|
||||
|
||||
static inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.
|
||||
static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }
|
||||
// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.
|
||||
static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }
|
||||
static inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
static inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
static inline void mh_chacha_block(const uint* x, uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
static inline void mh_cache_segment(__global uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
__global uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
static inline void mh_mixer(uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
static inline void mh_item(__global const uint* cache, uint t, uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
__global const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
static inline uint mh_word(__global const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.
|
||||
// The same constants as memhard.h in this pack (one emitter, three dialects).
|
||||
__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {
|
||||
uint seg = (uint)get_global_id(0);
|
||||
if (seg < nSegments) mh_cache_segment(cache, seg);
|
||||
}
|
||||
__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {
|
||||
uint t = (uint)get_global_id(0);
|
||||
if (t < nItems) {
|
||||
uint s[16];
|
||||
mh_item(cache, t, s);
|
||||
__global uint* d = ds + ((ulong)t * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
}
|
||||
|
||||
// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the
|
||||
// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and
|
||||
// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
static inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global uint* scratch, uint groups, uint salt) {
|
||||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ 0x67a9a7beu; x += 0x9e3779b9u; x = splitmix32(x); r0 = x ^ 0x1a155b25u; } // SEEDW[0], 0x9e3779b9u * 1u, SEEDW[1]
|
||||
{ uint x = nonce ^ 0x1a155b25u; x += 0x3c6ef372u; x = splitmix32(x); r1 = x ^ 0xfddfb732u; } // SEEDW[1], 0x9e3779b9u * 2u, SEEDW[2]
|
||||
{ uint x = nonce ^ 0xfddfb732u; x += 0xdaa66d2bu; x = splitmix32(x); r2 = x ^ 0x4b5af2e8u; } // SEEDW[2], 0x9e3779b9u * 3u, SEEDW[3]
|
||||
{ uint x = nonce ^ 0x4b5af2e8u; x += 0x78dde6e4u; x = splitmix32(x); r3 = x ^ 0xc55caf33u; } // SEEDW[3], 0x9e3779b9u * 4u, SEEDW[4]
|
||||
{ uint x = nonce ^ 0xc55caf33u; x += 0x1715609du; x = splitmix32(x); r4 = x ^ 0xa27c13b7u; } // SEEDW[4], 0x9e3779b9u * 5u, SEEDW[5]
|
||||
{ uint x = nonce ^ 0xa27c13b7u; x += 0xb54cda56u; x = splitmix32(x); r5 = x ^ 0x06628a48u; } // SEEDW[5], 0x9e3779b9u * 6u, SEEDW[6]
|
||||
{ uint x = nonce ^ 0x06628a48u; x += 0x5384540fu; x = splitmix32(x); r6 = x ^ 0x03852469u; } // SEEDW[6], 0x9e3779b9u * 7u, SEEDW[7]
|
||||
{ uint x = nonce ^ 0x03852469u; x += 0xf1bbcdc8u; x = splitmix32(x); r7 = x ^ 0x67a9a7beu; } // SEEDW[7], 0x9e3779b9u * 8u, SEEDW[0]
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
{ uint s_ = r7 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = mul_hi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint s_ = r5 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = mul_hi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
|
||||
#if IGNEUM_EXCHANGE != 0
|
||||
// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the
|
||||
// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a
|
||||
// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.
|
||||
IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {
|
||||
if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }
|
||||
}
|
||||
#endif
|
||||
|
||||
// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.
|
||||
IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw, __global uint* scratch, uint groups, uint salt) {
|
||||
uint lane = (uint)get_global_id(0) & 31u;
|
||||
uint warp_ = (uint)get_global_id(0) >> 5;
|
||||
uint nwarps_ = (uint)get_global_size(0) >> 5;
|
||||
__global uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint lid = (uint)get_local_id(0);
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];
|
||||
#if IGNEUM_EXCHANGE == 0
|
||||
IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);
|
||||
uint xk = 0u;
|
||||
#else
|
||||
(void)lid;
|
||||
#endif
|
||||
{ uint x = nonce ^ iw0; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw1; }
|
||||
{ uint x = nonce ^ iw1; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw2; }
|
||||
{ uint x = nonce ^ iw2; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw3; }
|
||||
{ uint x = nonce ^ iw3; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw4; }
|
||||
{ uint x = nonce ^ iw4; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw5; }
|
||||
{ uint x = nonce ^ iw5; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw6; }
|
||||
{ uint x = nonce ^ iw6; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw7; }
|
||||
{ uint x = nonce ^ iw7; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw0; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 4 scratch
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r3, 4u); r6 = r6 ^ t_; } // 6 shfl
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r5, 8u); r1 = r1 ^ t_; } // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r4, 2u); r3 = r3 ^ t_; } // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
{ uint t_; IGNEUM_SHFL_XOR(t_, r6, 8u); r0 = r0 ^ t_; } // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = mul_hi(r2, r5); // 22 mulhi
|
||||
{ uint s_ = r7 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 23 scratch
|
||||
r7 = mul_hi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = mul_hi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 37 scratch
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint s_ = r5 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 44 scratch
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = mul_hi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint s_ = r4 & 63u; uint4 v_ = vload4(s_, arena); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
136
proto-cuda/packs-readwidth/scr8k32/kernel_bound.cu
Normal file
136
proto-cuda/packs-readwidth/scr8k32/kernel_bound.cu
Normal file
|
|
@ -0,0 +1,136 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.
|
||||
// Host declarations (also in program_bound.h if present):
|
||||
// struct IgneumInitWords { uint32_t w[8]; };
|
||||
// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);
|
||||
// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);
|
||||
#include <cuda_runtime.h>
|
||||
#include <cstdint>
|
||||
#include "program.h"
|
||||
|
||||
struct IgneumInitWords { uint32_t w[8]; };
|
||||
|
||||
__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }
|
||||
__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
__device__ __forceinline__ uint32_t scr_fill(uint32_t gbase, uint32_t lane, uint32_t slot, uint32_t j) { uint32_t sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw, uint32_t* scratch, uint32_t groups, uint32_t salt) {
|
||||
uint32_t lane = (blockIdx.x * blockDim.x + threadIdx.x) & 31u;
|
||||
uint32_t warp_ = (blockIdx.x * blockDim.x + threadIdx.x) >> 5;
|
||||
uint32_t nwarps_ = (gridDim.x * blockDim.x) >> 5;
|
||||
uint32_t* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint32_t g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint32_t gid = g_ * 32u + lane;
|
||||
uint32_t gbase = baseNonce + g_ * 32u;
|
||||
uint32_t tag = salt + g_;
|
||||
uint32_t nonce = baseNonce + gid;
|
||||
uint32_t r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint32_t x = nonce ^ iw.w[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ iw.w[1]; }
|
||||
{ uint32_t x = nonce ^ iw.w[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ iw.w[2]; }
|
||||
{ uint32_t x = nonce ^ iw.w[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ iw.w[3]; }
|
||||
{ uint32_t x = nonce ^ iw.w[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ iw.w[4]; }
|
||||
{ uint32_t x = nonce ^ iw.w[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ iw.w[5]; }
|
||||
{ uint32_t x = nonce ^ iw.w[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ iw.w[6]; }
|
||||
{ uint32_t x = nonce ^ iw.w[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ iw.w[7]; }
|
||||
{ uint32_t x = nonce ^ iw.w[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ iw.w[0]; }
|
||||
|
||||
for (uint32_t it = 0u; it < 8u; ++it) {
|
||||
uint32_t sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0 mad
|
||||
r1 = r1 + r7 + ((((sel >> 4u) & 1u) != 0u) ? 0xc3bd2355u : 0x42da7657u); // 1 add
|
||||
r2 = r2 + r3 + ((((sel >> 26u) & 1u) != 0u) ? 0x2735a174u : 0x61f0b51cu); // 2 add
|
||||
r4 = r0 * r6 + r4; // 3 mad
|
||||
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4 scratch
|
||||
r4 = r4 ^ ds[r1 & mask]; // 5 load
|
||||
r6 = r6 ^ __shfl_xor_sync(0xffffffffu, r3, 4); // 6 shfl
|
||||
r1 = r1 ^ __shfl_xor_sync(0xffffffffu, r5, 8); // 7 shfl
|
||||
r7 = r7 ^ r5; // 8 xor
|
||||
r3 = r3 | r4; // 9 or
|
||||
r1 = r1 | r2; // 10 or
|
||||
r4 = r4 ^ ds[r3 & mask]; // 11 load
|
||||
r6 = r6 | r2; // 12 or
|
||||
r2 = r2 * r5; // 13 mul
|
||||
r1 = r1 ^ ds[r2 & mask]; // 14 load
|
||||
r7 = rotl_imm(r7, 1u); // 15 rotl
|
||||
{ uint32_t s_ = r6 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16 scratch
|
||||
r7 = r7 ^ ds[r4 & mask]; // 17 load
|
||||
r3 = r3 ^ __shfl_xor_sync(0xffffffffu, r4, 2); // 18 shfl
|
||||
r4 = r0 * r2 + r4; // 19 mad
|
||||
r0 = r0 ^ __shfl_xor_sync(0xffffffffu, r6, 8); // 20 shfl
|
||||
r5 = r5 ^ r7; // 21 xor
|
||||
r2 = __umulhi(r2, r5); // 22 mulhi
|
||||
{ uint32_t s_ = r7 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23 scratch
|
||||
r7 = __umulhi(r7, r3); // 24 mulhi
|
||||
r5 = r5 | r4; // 25 or
|
||||
r4 = r5 * r2 + r4; // 26 mad
|
||||
r5 = r5 * r1; // 27 mul
|
||||
r6 = __umulhi(r6, r7); // 28 mulhi
|
||||
r6 = r6 + r1 + ((((sel >> 9u) & 1u) != 0u) ? 0x187a9128u : 0x3b2d2124u); // 29 add
|
||||
r6 = rotr_var(r6, r7); // 30 rotr
|
||||
r3 = r3 ^ ds[r1 & mask]; // 31 load
|
||||
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32 scratch
|
||||
r0 = r0 + r4 + ((((sel >> 18u) & 1u) != 0u) ? 0x351dde38u : 0x2c35699fu); // 33 add
|
||||
{ uint32_t s_ = r2 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34 scratch
|
||||
r0 = r0 * r3; // 35 mul
|
||||
r2 = r2 ^ r5; // 36 xor
|
||||
{ uint32_t s_ = r0 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37 scratch
|
||||
r1 = r3 * r5 + r1; // 38 mad
|
||||
r0 = r0 + r3 + ((((sel >> 25u) & 1u) != 0u) ? 0x1b053acfu : 0xa907b90bu); // 39 add
|
||||
r2 = rotr_var(r2, r5); // 40 rotr
|
||||
r3 = r3 * r2; // 41 mul
|
||||
r1 = r1 + r5 + ((((sel >> 20u) & 1u) != 0u) ? 0x6058c2e3u : 0xa32e000cu); // 42 add
|
||||
r3 = r3 ^ r4; // 43 xor
|
||||
{ uint32_t s_ = r5 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44 scratch
|
||||
r1 = r1 + r5 + ((((sel >> 7u) & 1u) != 0u) ? 0x1907970cu : 0x81b8bc2cu); // 45 add
|
||||
r7 = r7 ^ r1; // 46 xor
|
||||
r0 = r0 + r3 + ((((sel >> 31u) & 1u) != 0u) ? 0x36360066u : 0x838b5065u); // 47 add
|
||||
r7 = __umulhi(r7, r5); // 48 mulhi
|
||||
r0 = r0 ^ ds[r2 & mask]; // 49 load
|
||||
r2 = r2 - r6; // 50 sub
|
||||
r7 = r7 - r5; // 51 sub
|
||||
r2 = r2 ^ r3; // 52 xor
|
||||
r7 = r7 - r0; // 53 sub
|
||||
r3 = r5 * r0 + r3; // 54 mad
|
||||
r7 = r7 ^ r5; // 55 xor
|
||||
r2 = r2 ^ ds[r7 & mask]; // 56 load
|
||||
r5 = r5 - r6; // 57 sub
|
||||
r1 = r1 ^ ds[r3 & mask]; // 58 load
|
||||
{ uint32_t s_ = r4 & 63u; uint4 v_ = *(const uint4*)(arena + s_ * 4u); uint32_t m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint32_t w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint32_t w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint32_t w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint32_t x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59 scratch
|
||||
r4 = r4 - r6; // 60 sub
|
||||
r1 = r1 * r2; // 61 mul
|
||||
r3 = r6 * r0 + r3; // 62 mad
|
||||
r0 = r0 + r1 + ((((sel >> 16u) & 1u) != 0u) ? 0xc88e2942u : 0x2fe0e98bu); // 63 add
|
||||
}
|
||||
uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;
|
||||
}
|
||||
}
|
||||
|
||||
// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).
|
||||
cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,
|
||||
IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {
|
||||
if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;
|
||||
uint32_t block = 32u * blockWarps;
|
||||
if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;
|
||||
igneum_hash_bound<<<warps / blockWarps, block>>>(ds, out, baseNonce, mask, iw, scratch, nonces / 32u, salt);
|
||||
return cudaGetLastError();
|
||||
}
|
||||
|
||||
cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {
|
||||
cudaFuncAttributes attr;
|
||||
cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);
|
||||
if (e != cudaSuccess) return e;
|
||||
*numRegs = attr.numRegs;
|
||||
return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);
|
||||
}
|
||||
108
proto-cuda/packs-readwidth/scr8k32/memhard.h
Normal file
108
proto-cuda/packs-readwidth/scr8k32/memhard.h
Normal file
|
|
@ -0,0 +1,108 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.
|
||||
// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).
|
||||
// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
#if defined(__CUDACC__)
|
||||
#define IGNEUM_HD __host__ __device__ __forceinline__
|
||||
#elif defined(_MSC_VER) && !defined(__cplusplus)
|
||||
#define IGNEUM_HD static __inline
|
||||
#else
|
||||
#define IGNEUM_HD static inline
|
||||
#endif
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
IGNEUM_HD uint32_t mh_rotl(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
IGNEUM_HD void mh_chacha_block(const uint32_t* x, uint32_t* y) {
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint32_t r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint32_t i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
IGNEUM_HD void mh_cache_segment(uint32_t* cache, uint32_t seg) {
|
||||
uint32_t prev[16]; uint32_t x[16]; uint32_t y[16];
|
||||
for (uint32_t i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint32_t j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
uint32_t* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
IGNEUM_HD void mh_mixer(uint32_t* s, uint32_t rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
IGNEUM_HD void mh_item(const uint32_t* cache, uint32_t t, uint32_t* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint32_t r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
const uint32_t* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint32_t i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
IGNEUM_HD uint32_t mh_word(const uint32_t* cache, uint32_t w) { uint32_t s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
106
proto-cuda/packs-readwidth/scr8k32/memhard.metal
Normal file
106
proto-cuda/packs-readwidth/scr8k32/memhard.metal
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
// Memory-hard dataset core (MEMHARD.md). Cache: 2^26 words in 2^16 segments of 64 chained ChaCha12 lines.
|
||||
// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.
|
||||
#define MH_CACHE_LINE_MASK 0x003fffffu
|
||||
#define MH_SEGMENT_LINES 64u
|
||||
#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }
|
||||
inline uint mh_rotl(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31 at every call site
|
||||
|
||||
// y = ChaCha12 core(x) + x
|
||||
inline void mh_chacha_block(const thread uint* x, thread uint* y) {
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] = x[i];
|
||||
for (uint r = 0u; r < 6u; ++r) {
|
||||
MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)
|
||||
MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)
|
||||
}
|
||||
for (uint i = 0u; i < 16u; ++i) y[i] += x[i];
|
||||
}
|
||||
|
||||
// One cache segment: 64 chained lines written at cache[seg * 1024]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.
|
||||
inline void mh_cache_segment(device uint* cache, uint seg) {
|
||||
uint prev[16]; uint x[16]; uint y[16];
|
||||
for (uint i = 0u; i < 16u; ++i) prev[i] = 0u;
|
||||
for (uint j = 0u; j < MH_SEGMENT_LINES; ++j) {
|
||||
x[0] = 0x61707865u ^ prev[0]; x[1] = 0x3320646eu ^ prev[1]; x[2] = 0x79622d32u ^ prev[2]; x[3] = 0x6b206574u ^ prev[3];
|
||||
x[4] = 0x3067619fu ^ prev[4];
|
||||
x[5] = 0x3c269176u ^ prev[5];
|
||||
x[6] = 0x84a03b03u ^ prev[6];
|
||||
x[7] = 0xf8c63294u ^ prev[7];
|
||||
x[8] = 0xff977c5bu ^ prev[8];
|
||||
x[9] = 0xe60def3eu ^ prev[9];
|
||||
x[10] = 0x63630141u ^ prev[10];
|
||||
x[11] = 0xb8fbcb58u ^ prev[11];
|
||||
x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = 0x49676e65u ^ prev[14]; x[15] = 0x756d4d48u ^ prev[15];
|
||||
mh_chacha_block(x, y);
|
||||
device uint* line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) { line[i] = y[i]; prev[i] = y[i]; }
|
||||
}
|
||||
}
|
||||
|
||||
// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.
|
||||
inline void mh_mixer(thread uint* s, uint rk) {
|
||||
s[0] = (s[0] ^ (0xbab68293u + rk)) * 0x42146205u;
|
||||
s[1] = (s[1] ^ (0xcc162340u + rk)) * 0x52cbe0fbu;
|
||||
s[2] = (s[2] ^ (0x6ce151ccu + rk)) * 0x7ecf4a03u;
|
||||
s[3] = (s[3] ^ (0xe62b8997u + rk)) * 0x6728907fu;
|
||||
s[4] = (s[4] ^ (0xc9c80297u + rk)) * 0xd81d9751u;
|
||||
s[5] = (s[5] ^ (0xf74a1654u + rk)) * 0x132952c3u;
|
||||
s[6] = (s[6] ^ (0x3d704af5u + rk)) * 0xf60de277u;
|
||||
s[7] = (s[7] ^ (0x3cf522b7u + rk)) * 0x05358035u;
|
||||
s[8] = (s[8] ^ (0x2b9cac04u + rk)) * 0xbaf6499du;
|
||||
s[9] = (s[9] ^ (0xa880ac10u + rk)) * 0xe4db9667u;
|
||||
s[10] = (s[10] ^ (0x13e5dd1du + rk)) * 0x3e98f45du;
|
||||
s[11] = (s[11] ^ (0x6fc3e233u + rk)) * 0xd0004eddu;
|
||||
s[12] = (s[12] ^ (0x2d83eeacu + rk)) * 0x2691630du;
|
||||
s[13] = (s[13] ^ (0x9006e8bfu + rk)) * 0x9beb3bcfu;
|
||||
s[14] = (s[14] ^ (0x2c4b5362u + rk)) * 0xab310379u;
|
||||
s[15] = (s[15] ^ (0x31b49ee2u + rk)) * 0x99cfb423u;
|
||||
MH_QR(s[0], s[4], s[8], s[12], 20u, 20u, 19u, 4u) MH_QR(s[1], s[5], s[9], s[13], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[2], s[6], s[10], s[14], 20u, 20u, 19u, 4u) MH_QR(s[3], s[7], s[11], s[15], 20u, 20u, 19u, 4u)
|
||||
MH_QR(s[0], s[5], s[10], s[15], 26u, 3u, 3u, 27u) MH_QR(s[1], s[6], s[11], s[12], 26u, 3u, 3u, 27u)
|
||||
MH_QR(s[2], s[7], s[8], s[13], 26u, 3u, 3u, 27u) MH_QR(s[3], s[4], s[9], s[14], 26u, 3u, 3u, 27u)
|
||||
}
|
||||
|
||||
// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); 8 rounds of mixer + cache line s[0] & mask; final mixer.
|
||||
inline void mh_item(device const uint* cache, uint t, thread uint* s) {
|
||||
s[0] = 0x3067619fu;
|
||||
s[1] = 0x3c269176u;
|
||||
s[2] = 0x84a03b03u;
|
||||
s[3] = 0xf8c63294u;
|
||||
s[4] = 0xff977c5bu;
|
||||
s[5] = 0xe60def3eu;
|
||||
s[6] = 0x63630141u;
|
||||
s[7] = 0xb8fbcb58u;
|
||||
s[8] = t * 0x42146205u + 0xbab68293u;
|
||||
s[9] = t * 0x52cbe0fbu + 0xcc162340u;
|
||||
s[10] = t * 0x7ecf4a03u + 0x6ce151ccu;
|
||||
s[11] = t * 0x6728907fu + 0xe62b8997u;
|
||||
s[12] = t * 0xd81d9751u + 0xc9c80297u;
|
||||
s[13] = t * 0x132952c3u + 0xf74a1654u;
|
||||
s[14] = t * 0xf60de277u + 0x3d704af5u;
|
||||
s[15] = t * 0x05358035u + 0x3cf522b7u;
|
||||
for (uint r = 0u; r < 8u; ++r) {
|
||||
mh_mixer(s, 0x9E3779B9u * (r + 1u));
|
||||
device const uint* line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);
|
||||
for (uint i = 0u; i < 16u; ++i) s[i] ^= line[i];
|
||||
}
|
||||
mh_mixer(s, 0x9E3779B9u * 9u);
|
||||
}
|
||||
// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.
|
||||
inline uint mh_word(device const uint* cache, uint w) { uint s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }
|
||||
|
||||
// One thread per segment (2^16 threads).
|
||||
kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {
|
||||
mh_cache_segment(cache, gid);
|
||||
}
|
||||
// One thread per 64-byte item (dataset words / 16 threads).
|
||||
kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],
|
||||
uint gid [[thread_position_in_grid]]) {
|
||||
uint s[16];
|
||||
mh_item(cache, gid, s);
|
||||
device uint* d = dataset + gid * 16u;
|
||||
for (uint i = 0u; i < 16u; ++i) d[i] = s[i];
|
||||
}
|
||||
|
|
@ -15,7 +15,7 @@
|
|||
#define IGNEUM_SEED_BYTES_HEX "69676e65756d2d67656e65736973"
|
||||
#define IGNEUM_GENERATOR 2
|
||||
#define IGNEUM_PROGRAM_ATTEMPT 0
|
||||
#define IGNEUM_PROGRAM_ID 0x2f0992e568f38dc1ull
|
||||
#define IGNEUM_PROGRAM_ID 0xe0d27cd155da1553ull
|
||||
#define IGNEUM_DAY_STRING "2026-10-03"
|
||||
#define IGNEUM_DAY_BYTES_HEX "6461792f323032362d31302d3033"
|
||||
#define IGNEUM_DAY0 0x3067619fu
|
||||
|
|
@ -30,20 +30,20 @@
|
|||
#define IGNEUM_OP_MIX "add=9 load=8 scratch=8 mad=7 xor=7 mul=5 sub=5 mulhi=4 or=4 shfl=4 rotr=2 rotl=1"
|
||||
// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
|
||||
// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
|
||||
#define IGNEUM_LOAD_CLASS "scr8"
|
||||
#define IGNEUM_LOAD_CLASS "scr8k32"
|
||||
#define IGNEUM_LOAD_SLOTS 16
|
||||
#define IGNEUM_LOAD_MIX { 100, 0, 0 }
|
||||
#define IGNEUM_LOAD_WIDTH_COUNTS { 8, 0, 0 } // loads of 4, 16, 64 bytes per program
|
||||
#define IGNEUM_BYTES_PER_HASH 256
|
||||
#define IGNEUM_FOLD_ROT 11
|
||||
#define IGNEUM_FOLD_MUL 0x9e3779b1u
|
||||
// Variant 5: persistent warps, a 1 MiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
// Variant 5: persistent warps, a 32 KiB scratch per launched warp (the host launches N warps and passes scratch,
|
||||
// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).
|
||||
#define IGNEUM_PERSISTENT_WARPS 1
|
||||
#define IGNEUM_SCRATCH_OPS 8 // scratch read-modify-writes per program (64 per hash)
|
||||
#define IGNEUM_SCRATCH_SLOTS 2048u
|
||||
#define IGNEUM_SCRATCH_WORDS_PER_LANE 8192u
|
||||
#define IGNEUM_SCRATCH_BYTES_PER_WARP 1048576u
|
||||
#define IGNEUM_SCRATCH_SLOTS 64u
|
||||
#define IGNEUM_SCRATCH_WORDS_PER_LANE 256u
|
||||
#define IGNEUM_SCRATCH_BYTES_PER_WARP 32768u
|
||||
// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)
|
||||
#define IGNEUM_DATASET_MODE 1
|
||||
|
||||
130
proto-cuda/packs-readwidth/scr8k32/program.json
Normal file
130
proto-cuda/packs-readwidth/scr8k32/program.json
Normal file
|
|
@ -0,0 +1,130 @@
|
|||
{
|
||||
"format": "igneum-program-pack-3",
|
||||
"generator": 2,
|
||||
"attempt": 0,
|
||||
"program_id": "0xe0d27cd155da1553",
|
||||
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
|
||||
"dataset_mode": "memory-hard",
|
||||
"seed": "igneum-genesis",
|
||||
"seed_bytes": "69676e65756d2d67656e65736973",
|
||||
"seed_words": ["0x67a9a7be", "0x1a155b25", "0xfddfb732", "0x4b5af2e8", "0xc55caf33", "0xa27c13b7", "0x06628a48", "0x03852469"],
|
||||
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
|
||||
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
|
||||
"lanes": 32,
|
||||
"registers": 8,
|
||||
"iterations": 8,
|
||||
"instruction_count": 64,
|
||||
"loads_per_hash": 128,
|
||||
"load_class": "scr8k32",
|
||||
"load_slots": 16,
|
||||
"load_mix_percent_4_16_64": [100, 0, 0],
|
||||
"load_width_counts_4_16_64": [8, 0, 0],
|
||||
"bytes_per_hash": 256,
|
||||
"scratch_ops_per_hash": 64,
|
||||
"scratch_kib_per_warp": 32,
|
||||
"scratch": "variant 5 (measurement only): persistent warps; a 32 KiB scratch per warp of 64 16-byte slots per lane (lane-major); slot = src & 0x3f; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)",
|
||||
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
|
||||
"op_mix": {"add": 9, "load": 8, "scratch": 8, "mad": 7, "xor": 7, "mul": 5, "sub": 5, "mulhi": 4, "or": 4, "shfl": 4, "rotr": 2, "rotl": 1},
|
||||
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
|
||||
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
|
||||
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
|
||||
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
|
||||
"op_semantics": {
|
||||
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
|
||||
"sub": "dst = dst - src",
|
||||
"mul": "dst = dst * src (low 32)",
|
||||
"mulhi": "dst = high 32 bits of dst * src",
|
||||
"xor": "dst = dst ^ src",
|
||||
"or": "dst = dst | src",
|
||||
"rotl": "dst = rotl(dst, rot), rot in 1..31",
|
||||
"rotr": "dst = rotr(dst, src & 31)",
|
||||
"mad": "dst = src * src2 + dst",
|
||||
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
|
||||
"load": "dst = dst ^ dataset[src & dataset.mask]",
|
||||
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
|
||||
},
|
||||
"dataset": {
|
||||
"log2_words": 28,
|
||||
"bytes": 1073741824,
|
||||
"mask": "0x0fffffff",
|
||||
"day": "2026-10-03",
|
||||
"day_bytes": "6461792f323032362d31302d3033",
|
||||
"day_words_from": "seed_words_from_bytes(day_bytes)",
|
||||
"d0": "0x3067619f",
|
||||
"d1": "0x3c269176",
|
||||
"mode": "memory-hard",
|
||||
"spec": "proto-metal/MEMHARD.md",
|
||||
"key": ["0x3067619f", "0x3c269176", "0x84a03b03", "0xf8c63294", "0xff977c5b", "0xe60def3e", "0x63630141", "0xb8fbcb58"],
|
||||
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
|
||||
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
|
||||
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [20, 20, 19, 4, 26, 3, 3, 27], "mul": ["0x42146205", "0x52cbe0fb", "0x7ecf4a03", "0x6728907f", "0xd81d9751", "0x132952c3", "0xf60de277", "0x05358035", "0xbaf6499d", "0xe4db9667", "0x3e98f45d", "0xd0004edd", "0x2691630d", "0x9beb3bcf", "0xab310379", "0x99cfb423"], "rc": ["0xbab68293", "0xcc162340", "0x6ce151cc", "0xe62b8997", "0xc9c80297", "0xf74a1654", "0x3d704af5", "0x3cf522b7", "0x2b9cac04", "0xa880ac10", "0x13e5dd1d", "0x6fc3e233", "0x2d83eeac", "0x9006e8bf", "0x2c4b5362", "0x31b49ee2"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
|
||||
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: s = M_r(s); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then s = M_8(s); item(t) = s",
|
||||
"word": "dataset[w] = item(w >> 4)[w & 15]"
|
||||
},
|
||||
"instructions": [
|
||||
{"i": 0, "op": "mad", "dst": 2, "src": 3, "src2": 4, "imm": "0xbf7b174d", "imm2": "0x337b762e", "rot": 17, "bit": 2, "mask": 2, "width": 1},
|
||||
{"i": 1, "op": "add", "dst": 1, "src": 7, "src2": 2, "imm": "0x42da7657", "imm2": "0xc3bd2355", "rot": 25, "bit": 4, "mask": 16, "width": 1},
|
||||
{"i": 2, "op": "add", "dst": 2, "src": 3, "src2": 0, "imm": "0x61f0b51c", "imm2": "0x2735a174", "rot": 4, "bit": 26, "mask": 2, "width": 1},
|
||||
{"i": 3, "op": "mad", "dst": 4, "src": 0, "src2": 6, "imm": "0x679648a8", "imm2": "0x3044ba32", "rot": 31, "bit": 31, "mask": 4, "width": 1},
|
||||
{"i": 4, "op": "scratch", "dst": 7, "src": 2, "src2": 6, "imm": "0x5d1ca2a2", "imm2": "0xe2481807", "rot": 24, "bit": 3, "mask": 1, "width": 1},
|
||||
{"i": 5, "op": "load", "dst": 4, "src": 1, "src2": 2, "imm": "0x987c017a", "imm2": "0xf4d60559", "rot": 2, "bit": 0, "mask": 4, "width": 1},
|
||||
{"i": 6, "op": "shfl", "dst": 6, "src": 3, "src2": 7, "imm": "0x6ea7b2df", "imm2": "0x9fce5071", "rot": 7, "bit": 15, "mask": 4, "width": 1},
|
||||
{"i": 7, "op": "shfl", "dst": 1, "src": 5, "src2": 1, "imm": "0x26a2ecde", "imm2": "0xfec6ad22", "rot": 15, "bit": 11, "mask": 8, "width": 1},
|
||||
{"i": 8, "op": "xor", "dst": 7, "src": 5, "src2": 2, "imm": "0xbe4b445c", "imm2": "0x17a5a9c7", "rot": 8, "bit": 8, "mask": 1, "width": 1},
|
||||
{"i": 9, "op": "or", "dst": 3, "src": 4, "src2": 1, "imm": "0xccb7d785", "imm2": "0xc335364c", "rot": 14, "bit": 12, "mask": 4, "width": 1},
|
||||
{"i": 10, "op": "or", "dst": 1, "src": 2, "src2": 3, "imm": "0x4e7dc10d", "imm2": "0x196d165c", "rot": 14, "bit": 27, "mask": 16, "width": 1},
|
||||
{"i": 11, "op": "load", "dst": 4, "src": 3, "src2": 1, "imm": "0xc5c3b55d", "imm2": "0xec061424", "rot": 26, "bit": 27, "mask": 8, "width": 1},
|
||||
{"i": 12, "op": "or", "dst": 6, "src": 2, "src2": 3, "imm": "0x306542fe", "imm2": "0x1bb1b429", "rot": 31, "bit": 0, "mask": 2, "width": 1},
|
||||
{"i": 13, "op": "mul", "dst": 2, "src": 5, "src2": 6, "imm": "0xa672cdd3", "imm2": "0x59a4829c", "rot": 22, "bit": 13, "mask": 16, "width": 1},
|
||||
{"i": 14, "op": "load", "dst": 1, "src": 2, "src2": 5, "imm": "0x028b4d37", "imm2": "0x7bbd78ea", "rot": 15, "bit": 2, "mask": 8, "width": 1},
|
||||
{"i": 15, "op": "rotl", "dst": 7, "src": 6, "src2": 6, "imm": "0x5c88a1a7", "imm2": "0x5c628769", "rot": 1, "bit": 3, "mask": 8, "width": 1},
|
||||
{"i": 16, "op": "scratch", "dst": 3, "src": 6, "src2": 7, "imm": "0xbac2ae81", "imm2": "0xcbbc7bdb", "rot": 18, "bit": 8, "mask": 8, "width": 1},
|
||||
{"i": 17, "op": "load", "dst": 7, "src": 4, "src2": 2, "imm": "0xe8ab93e9", "imm2": "0xa00de107", "rot": 2, "bit": 1, "mask": 16, "width": 1},
|
||||
{"i": 18, "op": "shfl", "dst": 3, "src": 4, "src2": 7, "imm": "0x1d2b8cab", "imm2": "0x80b4f9a2", "rot": 14, "bit": 25, "mask": 2, "width": 1},
|
||||
{"i": 19, "op": "mad", "dst": 4, "src": 0, "src2": 2, "imm": "0x5fba7bc2", "imm2": "0xdf099cfb", "rot": 4, "bit": 15, "mask": 16, "width": 1},
|
||||
{"i": 20, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0x0a3056de", "imm2": "0x7f0c25c3", "rot": 27, "bit": 13, "mask": 8, "width": 1},
|
||||
{"i": 21, "op": "xor", "dst": 5, "src": 7, "src2": 4, "imm": "0xbd066e1d", "imm2": "0x6d3ddc5a", "rot": 2, "bit": 29, "mask": 1, "width": 1},
|
||||
{"i": 22, "op": "mulhi", "dst": 2, "src": 5, "src2": 0, "imm": "0xc7e9887a", "imm2": "0x19ec898f", "rot": 14, "bit": 9, "mask": 1, "width": 1},
|
||||
{"i": 23, "op": "scratch", "dst": 3, "src": 7, "src2": 2, "imm": "0xc7fcfc8f", "imm2": "0x8528b94f", "rot": 17, "bit": 13, "mask": 4, "width": 1},
|
||||
{"i": 24, "op": "mulhi", "dst": 7, "src": 3, "src2": 5, "imm": "0xd91641e8", "imm2": "0xaf77faf2", "rot": 22, "bit": 21, "mask": 1, "width": 1},
|
||||
{"i": 25, "op": "or", "dst": 5, "src": 4, "src2": 0, "imm": "0x84c03868", "imm2": "0xf6c691b7", "rot": 29, "bit": 14, "mask": 8, "width": 1},
|
||||
{"i": 26, "op": "mad", "dst": 4, "src": 5, "src2": 2, "imm": "0x3bb2b6ba", "imm2": "0x49d95fd5", "rot": 1, "bit": 5, "mask": 8, "width": 1},
|
||||
{"i": 27, "op": "mul", "dst": 5, "src": 1, "src2": 7, "imm": "0x0a816217", "imm2": "0x405c4f73", "rot": 13, "bit": 27, "mask": 4, "width": 1},
|
||||
{"i": 28, "op": "mulhi", "dst": 6, "src": 7, "src2": 6, "imm": "0xd69c4715", "imm2": "0xe0ebc4ce", "rot": 29, "bit": 2, "mask": 8, "width": 1},
|
||||
{"i": 29, "op": "add", "dst": 6, "src": 1, "src2": 2, "imm": "0x3b2d2124", "imm2": "0x187a9128", "rot": 1, "bit": 9, "mask": 16, "width": 1},
|
||||
{"i": 30, "op": "rotr", "dst": 6, "src": 7, "src2": 0, "imm": "0x5c64a589", "imm2": "0x61c9a38d", "rot": 17, "bit": 21, "mask": 16, "width": 1},
|
||||
{"i": 31, "op": "load", "dst": 3, "src": 1, "src2": 7, "imm": "0xc37723fa", "imm2": "0xf3b024da", "rot": 16, "bit": 27, "mask": 16, "width": 1},
|
||||
{"i": 32, "op": "scratch", "dst": 1, "src": 0, "src2": 7, "imm": "0xcc7972c4", "imm2": "0xad098d15", "rot": 30, "bit": 21, "mask": 8, "width": 1},
|
||||
{"i": 33, "op": "add", "dst": 0, "src": 4, "src2": 4, "imm": "0x2c35699f", "imm2": "0x351dde38", "rot": 21, "bit": 18, "mask": 4, "width": 1},
|
||||
{"i": 34, "op": "scratch", "dst": 0, "src": 2, "src2": 3, "imm": "0xfae8902b", "imm2": "0x5cd8306f", "rot": 5, "bit": 28, "mask": 16, "width": 1},
|
||||
{"i": 35, "op": "mul", "dst": 0, "src": 3, "src2": 1, "imm": "0x4fa3f3db", "imm2": "0xdbf37e75", "rot": 7, "bit": 18, "mask": 4, "width": 1},
|
||||
{"i": 36, "op": "xor", "dst": 2, "src": 5, "src2": 5, "imm": "0x47f136c5", "imm2": "0x06ce9153", "rot": 19, "bit": 10, "mask": 2, "width": 1},
|
||||
{"i": 37, "op": "scratch", "dst": 4, "src": 0, "src2": 0, "imm": "0x04cc1d55", "imm2": "0x35c52d04", "rot": 11, "bit": 14, "mask": 2, "width": 1},
|
||||
{"i": 38, "op": "mad", "dst": 1, "src": 3, "src2": 5, "imm": "0x3958f280", "imm2": "0x8713c7e1", "rot": 5, "bit": 23, "mask": 16, "width": 1},
|
||||
{"i": 39, "op": "add", "dst": 0, "src": 3, "src2": 3, "imm": "0xa907b90b", "imm2": "0x1b053acf", "rot": 30, "bit": 25, "mask": 16, "width": 1},
|
||||
{"i": 40, "op": "rotr", "dst": 2, "src": 5, "src2": 4, "imm": "0xf8662282", "imm2": "0x10bb9e30", "rot": 8, "bit": 6, "mask": 2, "width": 1},
|
||||
{"i": 41, "op": "mul", "dst": 3, "src": 2, "src2": 4, "imm": "0x49087d74", "imm2": "0x6348b489", "rot": 17, "bit": 9, "mask": 16, "width": 1},
|
||||
{"i": 42, "op": "add", "dst": 1, "src": 5, "src2": 1, "imm": "0xa32e000c", "imm2": "0x6058c2e3", "rot": 25, "bit": 20, "mask": 8, "width": 1},
|
||||
{"i": 43, "op": "xor", "dst": 3, "src": 4, "src2": 2, "imm": "0x3dad0eb6", "imm2": "0xb97578cb", "rot": 3, "bit": 27, "mask": 1, "width": 1},
|
||||
{"i": 44, "op": "scratch", "dst": 3, "src": 5, "src2": 7, "imm": "0x374aec92", "imm2": "0x626f11df", "rot": 20, "bit": 18, "mask": 8, "width": 1},
|
||||
{"i": 45, "op": "add", "dst": 1, "src": 5, "src2": 6, "imm": "0x81b8bc2c", "imm2": "0x1907970c", "rot": 28, "bit": 7, "mask": 4, "width": 1},
|
||||
{"i": 46, "op": "xor", "dst": 7, "src": 1, "src2": 0, "imm": "0xef6ac348", "imm2": "0x963bb7e6", "rot": 26, "bit": 3, "mask": 8, "width": 1},
|
||||
{"i": 47, "op": "add", "dst": 0, "src": 3, "src2": 0, "imm": "0x838b5065", "imm2": "0x36360066", "rot": 3, "bit": 31, "mask": 4, "width": 1},
|
||||
{"i": 48, "op": "mulhi", "dst": 7, "src": 5, "src2": 0, "imm": "0x8458f7ac", "imm2": "0xc1c15026", "rot": 27, "bit": 15, "mask": 8, "width": 1},
|
||||
{"i": 49, "op": "load", "dst": 0, "src": 2, "src2": 4, "imm": "0x636a9dc4", "imm2": "0xac023d9b", "rot": 22, "bit": 29, "mask": 1, "width": 1},
|
||||
{"i": 50, "op": "sub", "dst": 2, "src": 6, "src2": 0, "imm": "0x2baec8c9", "imm2": "0x4390f156", "rot": 3, "bit": 12, "mask": 8, "width": 1},
|
||||
{"i": 51, "op": "sub", "dst": 7, "src": 5, "src2": 7, "imm": "0x19234061", "imm2": "0xe84dfade", "rot": 4, "bit": 19, "mask": 1, "width": 1},
|
||||
{"i": 52, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0xdc2cd71e", "imm2": "0x1b5d334b", "rot": 9, "bit": 8, "mask": 8, "width": 1},
|
||||
{"i": 53, "op": "sub", "dst": 7, "src": 0, "src2": 4, "imm": "0x605c31ec", "imm2": "0x9923ff88", "rot": 28, "bit": 25, "mask": 4, "width": 1},
|
||||
{"i": 54, "op": "mad", "dst": 3, "src": 5, "src2": 0, "imm": "0xc0cc51a6", "imm2": "0x3fd7701b", "rot": 20, "bit": 1, "mask": 4, "width": 1},
|
||||
{"i": 55, "op": "xor", "dst": 7, "src": 5, "src2": 5, "imm": "0xad7493e7", "imm2": "0x3e400372", "rot": 13, "bit": 8, "mask": 1, "width": 1},
|
||||
{"i": 56, "op": "load", "dst": 2, "src": 7, "src2": 1, "imm": "0x87e933c9", "imm2": "0x8c854c1b", "rot": 17, "bit": 3, "mask": 8, "width": 1},
|
||||
{"i": 57, "op": "sub", "dst": 5, "src": 6, "src2": 5, "imm": "0x11be3bc9", "imm2": "0xbbaa8e24", "rot": 6, "bit": 5, "mask": 16, "width": 1},
|
||||
{"i": 58, "op": "load", "dst": 1, "src": 3, "src2": 2, "imm": "0xa732351a", "imm2": "0xc01349cd", "rot": 14, "bit": 17, "mask": 16, "width": 1},
|
||||
{"i": 59, "op": "scratch", "dst": 1, "src": 4, "src2": 0, "imm": "0xb20547b2", "imm2": "0xc94655de", "rot": 27, "bit": 30, "mask": 1, "width": 1},
|
||||
{"i": 60, "op": "sub", "dst": 4, "src": 6, "src2": 7, "imm": "0x67cf904c", "imm2": "0x6873b216", "rot": 27, "bit": 7, "mask": 16, "width": 1},
|
||||
{"i": 61, "op": "mul", "dst": 1, "src": 2, "src2": 7, "imm": "0x93ab0bf4", "imm2": "0x96158375", "rot": 14, "bit": 0, "mask": 16, "width": 1},
|
||||
{"i": 62, "op": "mad", "dst": 3, "src": 6, "src2": 0, "imm": "0x41a443a3", "imm2": "0xe69d7919", "rot": 9, "bit": 0, "mask": 16, "width": 1},
|
||||
{"i": 63, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x2fe0e98b", "imm2": "0xc88e2942", "rot": 5, "bit": 16, "mask": 16, "width": 1}
|
||||
]
|
||||
}
|
||||
126
proto-cuda/packs-readwidth/scr8k32/program.metal
Normal file
126
proto-cuda/packs-readwidth/scr8k32/program.metal
Normal file
|
|
@ -0,0 +1,126 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
kernel void igneum_hash(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
device uint* scratch [[buffer(3)]],
|
||||
constant uint& groups [[buffer(4)]],
|
||||
constant uint& salt [[buffer(5)]],
|
||||
uint tid [[thread_position_in_grid]],
|
||||
uint nthreads [[threads_per_grid]]) {
|
||||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ SEEDW[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ SEEDW[1]; }
|
||||
{ uint x = nonce ^ SEEDW[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ SEEDW[2]; }
|
||||
{ uint x = nonce ^ SEEDW[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ SEEDW[3]; }
|
||||
{ uint x = nonce ^ SEEDW[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ SEEDW[4]; }
|
||||
{ uint x = nonce ^ SEEDW[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ SEEDW[5]; }
|
||||
{ uint x = nonce ^ SEEDW[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ SEEDW[6]; }
|
||||
{ uint x = nonce ^ SEEDW[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ SEEDW[7]; }
|
||||
{ uint x = nonce ^ SEEDW[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ SEEDW[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
|
||||
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
|
||||
r4 = r0 * r6 + r4; // 3
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
|
||||
r4 = r4 ^ dataset[r1 & MASK]; // 5
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
|
||||
r7 = r7 ^ r5; // 8
|
||||
r3 = r3 | r4; // 9
|
||||
r1 = r1 | r2; // 10
|
||||
r4 = r4 ^ dataset[r3 & MASK]; // 11
|
||||
r6 = r6 | r2; // 12
|
||||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
|
||||
r5 = r5 ^ r7; // 21
|
||||
r2 = mulhi(r2, r5); // 22
|
||||
{ uint s_ = r7 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
|
||||
r7 = mulhi(r7, r3); // 24
|
||||
r5 = r5 | r4; // 25
|
||||
r4 = r5 * r2 + r4; // 26
|
||||
r5 = r5 * r1; // 27
|
||||
r6 = mulhi(r6, r7); // 28
|
||||
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
|
||||
r6 = rotr_var(r6, r7); // 30
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
|
||||
r1 = r3 * r5 + r1; // 38
|
||||
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
|
||||
r2 = rotr_var(r2, r5); // 40
|
||||
r3 = r3 * r2; // 41
|
||||
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
|
||||
r3 = r3 ^ r4; // 43
|
||||
{ uint s_ = r5 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
|
||||
r7 = r7 ^ r1; // 46
|
||||
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
|
||||
r7 = mulhi(r7, r5); // 48
|
||||
r0 = r0 ^ dataset[r2 & MASK]; // 49
|
||||
r2 = r2 - r6; // 50
|
||||
r7 = r7 - r5; // 51
|
||||
r2 = r2 ^ r3; // 52
|
||||
r7 = r7 - r0; // 53
|
||||
r3 = r5 * r0 + r3; // 54
|
||||
r7 = r7 ^ r5; // 55
|
||||
r2 = r2 ^ dataset[r7 & MASK]; // 56
|
||||
r5 = r5 - r6; // 57
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 58
|
||||
{ uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
r4 = r4 - r6; // 60
|
||||
r1 = r1 * r2; // 61
|
||||
r3 = r6 * r0 + r3; // 62
|
||||
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
128
proto-cuda/packs-readwidth/scr8k32/program_bound.metal
Normal file
128
proto-cuda/packs-readwidth/scr8k32/program_bound.metal
Normal file
|
|
@ -0,0 +1,128 @@
|
|||
#include <metal_stdlib>
|
||||
using namespace metal;
|
||||
|
||||
#define MASK 0x0fffffffu
|
||||
constant uint SEEDW[8] = { 0x67a9a7beu, 0x1a155b25u, 0xfddfb732u, 0x4b5af2e8u, 0xc55caf33u, 0xa27c13b7u, 0x06628a48u, 0x03852469u };
|
||||
|
||||
inline uint splitmix32(uint x) {
|
||||
x ^= x >> 16; x *= 0x7feb352du;
|
||||
x ^= x >> 15; x *= 0x846ca68bu;
|
||||
x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31
|
||||
inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }
|
||||
inline uint ds_elem(uint i, uint d0, uint d1) {
|
||||
uint x = i ^ d0;
|
||||
x *= 0x9E3779B1u; x ^= x >> 15;
|
||||
x += d1;
|
||||
x *= 0x85EBCA77u; x ^= x >> 13;
|
||||
x *= 0xC2B2AE3Du; x ^= x >> 16;
|
||||
return x;
|
||||
}
|
||||
|
||||
// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a 32 KiB scratch per warp, 64 slots of
|
||||
// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not
|
||||
// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.
|
||||
inline uint scr_fill(uint gbase, uint lane, uint slot, uint j) { uint sw = (j == 0u) ? 0x67a9a7beu : ((j == 1u) ? 0x1a155b25u : 0xfddfb732u); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }
|
||||
// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.
|
||||
kernel void igneum_hash_bound(device const uint* dataset [[buffer(0)]],
|
||||
device ulong* out [[buffer(1)]],
|
||||
constant uint& baseNonce [[buffer(2)]],
|
||||
constant uint* initw [[buffer(3)]],
|
||||
device uint* scratch [[buffer(4)]],
|
||||
constant uint& groups [[buffer(5)]],
|
||||
constant uint& salt [[buffer(6)]],
|
||||
uint tid [[thread_position_in_grid]],
|
||||
uint nthreads [[threads_per_grid]]) {
|
||||
uint lane = tid & 31u;
|
||||
uint warp_ = tid >> 5;
|
||||
uint nwarps_ = nthreads >> 5;
|
||||
device uint* arena = scratch + ((size_t)warp_ * 32u + lane) * 256u;
|
||||
for (uint g_ = warp_; g_ < groups; g_ += nwarps_) {
|
||||
uint gid = g_ * 32u + lane;
|
||||
uint gbase = baseNonce + g_ * 32u;
|
||||
uint tag = salt + g_;
|
||||
uint nonce = baseNonce + gid;
|
||||
uint r0, r1, r2, r3, r4, r5, r6, r7;
|
||||
{ uint x = nonce ^ initw[0]; x += 0x9e3779b9u * 1u; x = splitmix32(x); r0 = x ^ initw[1]; }
|
||||
{ uint x = nonce ^ initw[1]; x += 0x9e3779b9u * 2u; x = splitmix32(x); r1 = x ^ initw[2]; }
|
||||
{ uint x = nonce ^ initw[2]; x += 0x9e3779b9u * 3u; x = splitmix32(x); r2 = x ^ initw[3]; }
|
||||
{ uint x = nonce ^ initw[3]; x += 0x9e3779b9u * 4u; x = splitmix32(x); r3 = x ^ initw[4]; }
|
||||
{ uint x = nonce ^ initw[4]; x += 0x9e3779b9u * 5u; x = splitmix32(x); r4 = x ^ initw[5]; }
|
||||
{ uint x = nonce ^ initw[5]; x += 0x9e3779b9u * 6u; x = splitmix32(x); r5 = x ^ initw[6]; }
|
||||
{ uint x = nonce ^ initw[6]; x += 0x9e3779b9u * 7u; x = splitmix32(x); r6 = x ^ initw[7]; }
|
||||
{ uint x = nonce ^ initw[7]; x += 0x9e3779b9u * 8u; x = splitmix32(x); r7 = x ^ initw[0]; }
|
||||
|
||||
for (uint it = 0u; it < 8u; ++it) {
|
||||
uint sel = r0;
|
||||
r2 = r3 * r4 + r2; // 0
|
||||
r1 = r1 + r7 + select(0x42da7657u, 0xc3bd2355u, ((sel >> 4u) & 1u) != 0u); // 1
|
||||
r2 = r2 + r3 + select(0x61f0b51cu, 0x2735a174u, ((sel >> 26u) & 1u) != 0u); // 2
|
||||
r4 = r0 * r6 + r4; // 3
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r7 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r7 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 4
|
||||
r4 = r4 ^ dataset[r1 & MASK]; // 5
|
||||
r6 = r6 ^ simd_shuffle_xor(r3, (ushort)4); // 6
|
||||
r1 = r1 ^ simd_shuffle_xor(r5, (ushort)8); // 7
|
||||
r7 = r7 ^ r5; // 8
|
||||
r3 = r3 | r4; // 9
|
||||
r1 = r1 | r2; // 10
|
||||
r4 = r4 ^ dataset[r3 & MASK]; // 11
|
||||
r6 = r6 | r2; // 12
|
||||
r2 = r2 * r5; // 13
|
||||
r1 = r1 ^ dataset[r2 & MASK]; // 14
|
||||
r7 = rotl_imm(r7, 1u); // 15
|
||||
{ uint s_ = r6 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 16
|
||||
r7 = r7 ^ dataset[r4 & MASK]; // 17
|
||||
r3 = r3 ^ simd_shuffle_xor(r4, (ushort)2); // 18
|
||||
r4 = r0 * r2 + r4; // 19
|
||||
r0 = r0 ^ simd_shuffle_xor(r6, (ushort)8); // 20
|
||||
r5 = r5 ^ r7; // 21
|
||||
r2 = mulhi(r2, r5); // 22
|
||||
{ uint s_ = r7 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 23
|
||||
r7 = mulhi(r7, r3); // 24
|
||||
r5 = r5 | r4; // 25
|
||||
r4 = r5 * r2 + r4; // 26
|
||||
r5 = r5 * r1; // 27
|
||||
r6 = mulhi(r6, r7); // 28
|
||||
r6 = r6 + r1 + select(0x3b2d2124u, 0x187a9128u, ((sel >> 9u) & 1u) != 0u); // 29
|
||||
r6 = rotr_var(r6, r7); // 30
|
||||
r3 = r3 ^ dataset[r1 & MASK]; // 31
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 32
|
||||
r0 = r0 + r4 + select(0x2c35699fu, 0x351dde38u, ((sel >> 18u) & 1u) != 0u); // 33
|
||||
{ uint s_ = r2 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r0 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r0 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 34
|
||||
r0 = r0 * r3; // 35
|
||||
r2 = r2 ^ r5; // 36
|
||||
{ uint s_ = r0 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r4 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r4 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 37
|
||||
r1 = r3 * r5 + r1; // 38
|
||||
r0 = r0 + r3 + select(0xa907b90bu, 0x1b053acfu, ((sel >> 25u) & 1u) != 0u); // 39
|
||||
r2 = rotr_var(r2, r5); // 40
|
||||
r3 = r3 * r2; // 41
|
||||
r1 = r1 + r5 + select(0xa32e000cu, 0x6058c2e3u, ((sel >> 20u) & 1u) != 0u); // 42
|
||||
r3 = r3 ^ r4; // 43
|
||||
{ uint s_ = r5 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r3 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r3 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 44
|
||||
r1 = r1 + r5 + select(0x81b8bc2cu, 0x1907970cu, ((sel >> 7u) & 1u) != 0u); // 45
|
||||
r7 = r7 ^ r1; // 46
|
||||
r0 = r0 + r3 + select(0x838b5065u, 0x36360066u, ((sel >> 31u) & 1u) != 0u); // 47
|
||||
r7 = mulhi(r7, r5); // 48
|
||||
r0 = r0 ^ dataset[r2 & MASK]; // 49
|
||||
r2 = r2 - r6; // 50
|
||||
r7 = r7 - r5; // 51
|
||||
r2 = r2 ^ r3; // 52
|
||||
r7 = r7 - r0; // 53
|
||||
r3 = r5 * r0 + r3; // 54
|
||||
r7 = r7 ^ r5; // 55
|
||||
r2 = r2 ^ dataset[r7 & MASK]; // 56
|
||||
r5 = r5 - r6; // 57
|
||||
r1 = r1 ^ dataset[r3 & MASK]; // 58
|
||||
{ uint s_ = r4 & 63u; uint4 v_ = *(device const uint4*)(arena + s_ * 4u); uint m_ = (v_.x == tag) ? 0xffffffffu : 0u; uint w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); uint w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); uint w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); uint x_ = r1 ^ w0_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w1_; x_ = (rotl_imm(x_, 11u) * 0x9e3779b1u) ^ w2_; r1 = x_; *(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); } // 59
|
||||
r4 = r4 - r6; // 60
|
||||
r1 = r1 * r2; // 61
|
||||
r3 = r6 * r0 + r3; // 62
|
||||
r0 = r0 + r1 + select(0x2fe0e98bu, 0xc88e2942u, ((sel >> 16u) & 1u) != 0u); // 63
|
||||
}
|
||||
uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);
|
||||
uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);
|
||||
out[gid] = ((ulong)hi << 32) | (ulong)lo;
|
||||
}
|
||||
}
|
||||
57
proto-cuda/packs-readwidth/scr8k32/vectors.h
Normal file
57
proto-cuda/packs-readwidth/scr8k32/vectors.h
Normal file
|
|
@ -0,0 +1,57 @@
|
|||
// Generated by igneum-pow export (generator v2) for seed "igneum-genesis". Do not edit by hand.
|
||||
// Expected outputs: igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset
|
||||
#pragma once
|
||||
#ifdef __cplusplus
|
||||
#include <cstdint>
|
||||
#else
|
||||
#include <stdint.h>
|
||||
#endif
|
||||
|
||||
#define IGNEUM_VEC_WARPS 3
|
||||
static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = { 0u, 4096u, 1000000u };
|
||||
static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {
|
||||
{ // base nonce 0
|
||||
0x82097370c10ee4eaull, 0xa0f16965a422e758ull, 0x2b7513e2593b170cull, 0x8fbb69f8974c4533ull, 0x500c384970d19f32ull, 0x72fb96e1820fa2b8ull, 0xca1521df6a934a65ull, 0x7407397e10548cbcull,
|
||||
0xa619f127d1aa896bull, 0x36ce1104c1517d86ull, 0x6078ddd46d6d26b2ull, 0x2d22ded89010fbd3ull, 0xb030b743355bf6f9ull, 0x8280c8c1f4946ef7ull, 0xbfdae5712c7b9ad4ull, 0x50e9136a89e2e046ull,
|
||||
0xfe0230d874a94f3aull, 0x8b6dbb8ff0546c32ull, 0x41438f3ab2d35cbcull, 0x6faf2dd5c5997f9cull, 0x52c3aae40019a0abull, 0xc455bf9926661472ull, 0x4eaf9e34af6c7c22ull, 0x9dcd05c3a3991115ull,
|
||||
0x3a4f0ed406a66796ull, 0x7615703c08eabd8aull, 0x6007bcf34b8c6341ull, 0x1fb11900437441aeull, 0x540cff895991c6acull, 0x367a10121393f054ull, 0xf0dd7aa43fbdc2f4ull, 0x53d26f9c120f8a64ull
|
||||
},
|
||||
{ // base nonce 4096
|
||||
0xc04adc4c89d1593bull, 0xc938ee393ee383dcull, 0xaef535444b6968fbull, 0x83581d97cebb4b30ull, 0x365315d0762bf586ull, 0xc26dfc2c40131920ull, 0x2f560c6fa351ead5ull, 0xc235996f6596d536ull,
|
||||
0x1adf3384c9f23712ull, 0x0c87a8ad0dc872c3ull, 0x82471bfd2a3182d4ull, 0xe8828b0fb7560877ull, 0x3fd9044dd153dd7full, 0x3a91a9618ce2c525ull, 0xf8c503abac8481f9ull, 0x56b0989f6014f8dbull,
|
||||
0xa1107a2732c588baull, 0xfc47a39e530d7efaull, 0x9237f7fc01727703ull, 0x01b5cbb25e6629c1ull, 0xae8e66fcec949f3eull, 0xb31655fbd75d3dbdull, 0x5a7750b585b3d0f1ull, 0x72f0484703cf30a0ull,
|
||||
0x4cea6de3f76e970bull, 0x64fd914fb1fb5096ull, 0xbdcca26f07e7182cull, 0xc86ba79905164a5bull, 0x521b7d3ac36f06cbull, 0x6b8618a80cef2e71ull, 0x13df43bef1f7852full, 0x202e92ce7b597a33ull
|
||||
},
|
||||
{ // base nonce 1000000
|
||||
0x4db5bf37f24811f2ull, 0x5e93450594a45e5full, 0xebe57bb921f163aaull, 0x0457e6f5ac702b1eull, 0x442d8926fceec0d4ull, 0x417256d19cce43d3ull, 0x3d61563a8303daceull, 0x4a43c5efac6f6d47ull,
|
||||
0x9ef8a8d0a98f5c7eull, 0xbee88175926bb251ull, 0xb3b8e73f1e427be1ull, 0x515405b57446beecull, 0xba4f1765e616bf9aull, 0x148e5d9895c48299ull, 0x303fed05bdcdd7d1ull, 0x44d07bf31dba804full,
|
||||
0x8616fd225f3851beull, 0x426a7a80ac1b462full, 0x6e0163361c5ec30full, 0x065b3666feb8d0e5ull, 0xf9bc697886c9983eull, 0x90bc61f358b511f4ull, 0xfa47afe197811c66ull, 0x12a39e4e67aa2e97ull,
|
||||
0xde01f49ccab26a42ull, 0x2c6e837e74897413ull, 0x62d22c8acdeb1d09ull, 0x7fa8da035f65bb0bull, 0xdc4ce47ce0d48b6bull, 0x6199717653754041ull, 0x3a5e113c0d160d86ull, 0x718b3357f2391b60ull
|
||||
}
|
||||
};
|
||||
|
||||
// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] (268435455).
|
||||
static const uint32_t IGNEUM_DS_HEAD[16] = {
|
||||
0xffc3cd94u, 0x5920ccd8u, 0x392f44bbu, 0x5e57f67au, 0x2f2bc2a9u, 0x620b0e36u, 0xbdc09014u, 0x436654bfu,
|
||||
0x311e0b48u, 0x1abd93adu, 0x59cc7ce8u, 0xee5247b2u, 0x86171fe8u, 0x6d874751u, 0xc9f7728fu, 0x7c2a435du
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_LAST_INDEX = 268435455u;
|
||||
static const uint32_t IGNEUM_DS_LAST = 0xa33ada72u;
|
||||
// 64 sampled dataset words (index, value) computed on the Mac.
|
||||
#define IGNEUM_DS_SAMPLES 64
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {
|
||||
59471966u, 217795994u, 208353206u, 42483309u, 172547758u, 148076330u, 183853158u, 214389424u, 267488061u, 169781097u, 184093494u, 153880993u, 84977930u, 46426879u, 3093825u, 225364072u, 44593546u, 260713159u, 168250303u, 52384140u, 223401610u, 45554030u, 95410555u, 175039924u, 79171087u, 267580473u, 24168642u, 37981670u, 171551130u, 195559979u, 204611762u, 140997658u, 138925853u, 86637313u, 20736778u, 219665210u, 160430336u, 264654675u, 8013395u, 228945585u, 213884386u, 104419827u, 44185464u, 142737231u, 99284897u, 132475900u, 61861762u, 132056166u, 262388043u, 91878046u, 117353561u, 124768597u, 71352993u, 190698941u, 46055428u, 55281366u, 165145231u, 106810753u, 171985651u, 232085256u, 159510492u, 40072060u, 209107596u, 39023794u
|
||||
};
|
||||
static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {
|
||||
0xe8b73d94u, 0x337028b5u, 0xafe148c9u, 0xab99f7aeu, 0x434ea619u, 0xd85cb880u, 0x54764c7fu, 0x82c7e420u, 0xedf4cb9eu, 0x9884c959u, 0x223ee793u, 0x3a9ccf69u, 0x81da4fd2u, 0xd6ce8cb9u, 0xe3922dcau, 0x3e7e6bdeu, 0x382a3acau, 0x567e7f7fu, 0x25a0f084u, 0xbfeef128u, 0xe338abfbu, 0x7c3b5280u, 0x909bc5f1u, 0xd8b74b9cu, 0x8e31a22eu, 0x26b5f1d8u, 0x79122c00u, 0xcafc3340u, 0xd5e02ea3u, 0x1aee1afdu, 0xdb090d9au, 0xb049f435u, 0x4954d8bau, 0x03797ba0u, 0x196eefbdu, 0xd153412au, 0xbe5d2c4bu, 0xdaa14f0eu, 0x8e61ed07u, 0x9e9a64c6u, 0x2e29ff36u, 0x392a8589u, 0xb56a5912u, 0xfa6e8b57u, 0xd1a737cbu, 0xb0fa841au, 0xbe1c341fu, 0xe25be0f1u, 0xe937f543u, 0xebab2248u, 0x8e1b607au, 0x202a2fedu, 0x95e2819cu, 0x9c9652d4u, 0x32fedef0u, 0xdecfff82u, 0xcb5d43e5u, 0xb735806au, 0x8905939cu, 0xfbf8472du, 0xada74e5du, 0x7ebdeeeau, 0x0119f2b3u, 0xa9a376b8u
|
||||
};
|
||||
// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^26 words.
|
||||
static const uint32_t IGNEUM_CACHE_HEAD[16] = {
|
||||
0x355a86d2u, 0x7957db1cu, 0xd21772afu, 0x6fc1e09bu, 0xd55ce61du, 0x6e6a278bu, 0xd3f543ceu, 0x223d8e82u,
|
||||
0x143ab337u, 0x2e9f05bdu, 0x2eb389bfu, 0x0c6e449eu, 0x5cfa4222u, 0xba6560feu, 0x8e3e1aa4u, 0xdbcc1d53u
|
||||
};
|
||||
static const uint32_t IGNEUM_CACHE_LAST[16] = {
|
||||
0x41190d91u, 0xbd277957u, 0x22ddbb49u, 0x6986f207u, 0xdf69a4d6u, 0x26401a3au, 0x818230fbu, 0xc417122du,
|
||||
0x3597b211u, 0xb553ce55u, 0xcf39cc0du, 0x3b7fc43au, 0x3fd43b00u, 0x67e1c80eu, 0xffa7ea7du, 0xca2960abu
|
||||
};
|
||||
static const uint64_t IGNEUM_CACHE_FNV64 = 0x48c4f5bf24166b2eull;
|
||||
36
proto-cuda/packs-readwidth/scr8k32/vectors.json
Normal file
36
proto-cuda/packs-readwidth/scr8k32/vectors.json
Normal file
|
|
@ -0,0 +1,36 @@
|
|||
{
|
||||
"seed": "igneum-genesis",
|
||||
"day": "2026-10-03",
|
||||
"dataset_mode": "memory-hard",
|
||||
"dataset_log2_words": 28,
|
||||
"mask": "0x0fffffff",
|
||||
"lanes": 32,
|
||||
"source": "igneum-pow (Rust) CPU interpreter, generator v2, memory-hard dataset",
|
||||
"warps": [
|
||||
{"base_nonce": 0, "expected": [
|
||||
"0x82097370c10ee4ea", "0xa0f16965a422e758", "0x2b7513e2593b170c", "0x8fbb69f8974c4533", "0x500c384970d19f32", "0x72fb96e1820fa2b8", "0xca1521df6a934a65", "0x7407397e10548cbc",
|
||||
"0xa619f127d1aa896b", "0x36ce1104c1517d86", "0x6078ddd46d6d26b2", "0x2d22ded89010fbd3", "0xb030b743355bf6f9", "0x8280c8c1f4946ef7", "0xbfdae5712c7b9ad4", "0x50e9136a89e2e046",
|
||||
"0xfe0230d874a94f3a", "0x8b6dbb8ff0546c32", "0x41438f3ab2d35cbc", "0x6faf2dd5c5997f9c", "0x52c3aae40019a0ab", "0xc455bf9926661472", "0x4eaf9e34af6c7c22", "0x9dcd05c3a3991115",
|
||||
"0x3a4f0ed406a66796", "0x7615703c08eabd8a", "0x6007bcf34b8c6341", "0x1fb11900437441ae", "0x540cff895991c6ac", "0x367a10121393f054", "0xf0dd7aa43fbdc2f4", "0x53d26f9c120f8a64"
|
||||
]},
|
||||
{"base_nonce": 4096, "expected": [
|
||||
"0xc04adc4c89d1593b", "0xc938ee393ee383dc", "0xaef535444b6968fb", "0x83581d97cebb4b30", "0x365315d0762bf586", "0xc26dfc2c40131920", "0x2f560c6fa351ead5", "0xc235996f6596d536",
|
||||
"0x1adf3384c9f23712", "0x0c87a8ad0dc872c3", "0x82471bfd2a3182d4", "0xe8828b0fb7560877", "0x3fd9044dd153dd7f", "0x3a91a9618ce2c525", "0xf8c503abac8481f9", "0x56b0989f6014f8db",
|
||||
"0xa1107a2732c588ba", "0xfc47a39e530d7efa", "0x9237f7fc01727703", "0x01b5cbb25e6629c1", "0xae8e66fcec949f3e", "0xb31655fbd75d3dbd", "0x5a7750b585b3d0f1", "0x72f0484703cf30a0",
|
||||
"0x4cea6de3f76e970b", "0x64fd914fb1fb5096", "0xbdcca26f07e7182c", "0xc86ba79905164a5b", "0x521b7d3ac36f06cb", "0x6b8618a80cef2e71", "0x13df43bef1f7852f", "0x202e92ce7b597a33"
|
||||
]},
|
||||
{"base_nonce": 1000000, "expected": [
|
||||
"0x4db5bf37f24811f2", "0x5e93450594a45e5f", "0xebe57bb921f163aa", "0x0457e6f5ac702b1e", "0x442d8926fceec0d4", "0x417256d19cce43d3", "0x3d61563a8303dace", "0x4a43c5efac6f6d47",
|
||||
"0x9ef8a8d0a98f5c7e", "0xbee88175926bb251", "0xb3b8e73f1e427be1", "0x515405b57446beec", "0xba4f1765e616bf9a", "0x148e5d9895c48299", "0x303fed05bdcdd7d1", "0x44d07bf31dba804f",
|
||||
"0x8616fd225f3851be", "0x426a7a80ac1b462f", "0x6e0163361c5ec30f", "0x065b3666feb8d0e5", "0xf9bc697886c9983e", "0x90bc61f358b511f4", "0xfa47afe197811c66", "0x12a39e4e67aa2e97",
|
||||
"0xde01f49ccab26a42", "0x2c6e837e74897413", "0x62d22c8acdeb1d09", "0x7fa8da035f65bb0b", "0xdc4ce47ce0d48b6b", "0x6199717653754041", "0x3a5e113c0d160d86", "0x718b3357f2391b60"
|
||||
]}
|
||||
],
|
||||
"dataset_head": ["0xffc3cd94", "0x5920ccd8", "0x392f44bb", "0x5e57f67a", "0x2f2bc2a9", "0x620b0e36", "0xbdc09014", "0x436654bf", "0x311e0b48", "0x1abd93ad", "0x59cc7ce8", "0xee5247b2", "0x86171fe8", "0x6d874751", "0xc9f7728f", "0x7c2a435d"],
|
||||
"dataset_last_index": 268435455,
|
||||
"dataset_last": "0xa33ada72",
|
||||
"dataset_samples": [{"index": 59471966, "value": "0xe8b73d94"}, {"index": 217795994, "value": "0x337028b5"}, {"index": 208353206, "value": "0xafe148c9"}, {"index": 42483309, "value": "0xab99f7ae"}, {"index": 172547758, "value": "0x434ea619"}, {"index": 148076330, "value": "0xd85cb880"}, {"index": 183853158, "value": "0x54764c7f"}, {"index": 214389424, "value": "0x82c7e420"}, {"index": 267488061, "value": "0xedf4cb9e"}, {"index": 169781097, "value": "0x9884c959"}, {"index": 184093494, "value": "0x223ee793"}, {"index": 153880993, "value": "0x3a9ccf69"}, {"index": 84977930, "value": "0x81da4fd2"}, {"index": 46426879, "value": "0xd6ce8cb9"}, {"index": 3093825, "value": "0xe3922dca"}, {"index": 225364072, "value": "0x3e7e6bde"}, {"index": 44593546, "value": "0x382a3aca"}, {"index": 260713159, "value": "0x567e7f7f"}, {"index": 168250303, "value": "0x25a0f084"}, {"index": 52384140, "value": "0xbfeef128"}, {"index": 223401610, "value": "0xe338abfb"}, {"index": 45554030, "value": "0x7c3b5280"}, {"index": 95410555, "value": "0x909bc5f1"}, {"index": 175039924, "value": "0xd8b74b9c"}, {"index": 79171087, "value": "0x8e31a22e"}, {"index": 267580473, "value": "0x26b5f1d8"}, {"index": 24168642, "value": "0x79122c00"}, {"index": 37981670, "value": "0xcafc3340"}, {"index": 171551130, "value": "0xd5e02ea3"}, {"index": 195559979, "value": "0x1aee1afd"}, {"index": 204611762, "value": "0xdb090d9a"}, {"index": 140997658, "value": "0xb049f435"}, {"index": 138925853, "value": "0x4954d8ba"}, {"index": 86637313, "value": "0x03797ba0"}, {"index": 20736778, "value": "0x196eefbd"}, {"index": 219665210, "value": "0xd153412a"}, {"index": 160430336, "value": "0xbe5d2c4b"}, {"index": 264654675, "value": "0xdaa14f0e"}, {"index": 8013395, "value": "0x8e61ed07"}, {"index": 228945585, "value": "0x9e9a64c6"}, {"index": 213884386, "value": "0x2e29ff36"}, {"index": 104419827, "value": "0x392a8589"}, {"index": 44185464, "value": "0xb56a5912"}, {"index": 142737231, "value": "0xfa6e8b57"}, {"index": 99284897, "value": "0xd1a737cb"}, {"index": 132475900, "value": "0xb0fa841a"}, {"index": 61861762, "value": "0xbe1c341f"}, {"index": 132056166, "value": "0xe25be0f1"}, {"index": 262388043, "value": "0xe937f543"}, {"index": 91878046, "value": "0xebab2248"}, {"index": 117353561, "value": "0x8e1b607a"}, {"index": 124768597, "value": "0x202a2fed"}, {"index": 71352993, "value": "0x95e2819c"}, {"index": 190698941, "value": "0x9c9652d4"}, {"index": 46055428, "value": "0x32fedef0"}, {"index": 55281366, "value": "0xdecfff82"}, {"index": 165145231, "value": "0xcb5d43e5"}, {"index": 106810753, "value": "0xb735806a"}, {"index": 171985651, "value": "0x8905939c"}, {"index": 232085256, "value": "0xfbf8472d"}, {"index": 159510492, "value": "0xada74e5d"}, {"index": 40072060, "value": "0x7ebdeeea"}, {"index": 209107596, "value": "0x0119f2b3"}, {"index": 39023794, "value": "0xa9a376b8"}],
|
||||
"cache_head": ["0x355a86d2", "0x7957db1c", "0xd21772af", "0x6fc1e09b", "0xd55ce61d", "0x6e6a278b", "0xd3f543ce", "0x223d8e82", "0x143ab337", "0x2e9f05bd", "0x2eb389bf", "0x0c6e449e", "0x5cfa4222", "0xba6560fe", "0x8e3e1aa4", "0xdbcc1d53"],
|
||||
"cache_last_line": ["0x41190d91", "0xbd277957", "0x22ddbb49", "0x6986f207", "0xdf69a4d6", "0x26401a3a", "0x818230fb", "0xc417122d", "0x3597b211", "0xb553ce55", "0xcf39cc0d", "0x3b7fc43a", "0x3fd43b00", "0x67e1c80e", "0xffa7ea7d", "0xca2960ab"],
|
||||
"cache_fnv1a64": "0x48c4f5bf24166b2e"
|
||||
}
|
||||
209
proto-metal/packbench.swift
Normal file
209
proto-metal/packbench.swift
Normal file
|
|
@ -0,0 +1,209 @@
|
|||
// packbench: runs a program pack (igneum-pow export) on the Mac's Metal GPU from its files alone: memhard.metal (cache
|
||||
// fill, dataset build), program.metal (igneum_hash), program.h (constants), vectors.json (the CPU reference's vectors).
|
||||
// Read-width experiment, 5 October 2026 (docs/plans/read-width.md): the Swift bench generates its own programs and does
|
||||
// not know the experiment's load classes; this harness runs whatever text the Rust emitter wrote, so Metal is checked
|
||||
// against the Rust CPU reference and timed without a Swift mirror of the generator. One file, no packages.
|
||||
//
|
||||
// swiftc -O -target arm64-apple-macos11 -o packbench packbench.swift -framework Metal
|
||||
// ./packbench --pack <dir> [--batches 5] [--batch-log2 24] [--group 256] [--warps 2048]
|
||||
//
|
||||
// Prints one RESULT line per run: vectors, cache and dataset checks, the batch fingerprint (FNV-1a 64 over the 2^B
|
||||
// outputs at base nonce 0) and MH/s by wall and by GPU time. Variant 5 packs (IGNEUM_PERSISTENT_WARPS) are launched
|
||||
// as --warps persistent warps with a 1 MiB scratch each; the batch is rounded to a multiple of 32 x warps.
|
||||
import Foundation
|
||||
import Metal
|
||||
|
||||
func nowMs() -> Double { return Double(DispatchTime.now().uptimeNanoseconds) / 1e6 }
|
||||
func fail(_ m: String) -> Never { print("FAIL: \(m)"); exit(1) }
|
||||
|
||||
struct Opts { var pack = ""; var batches = 5; var batchLog2 = 24; var group = 256; var warps = 2048 }
|
||||
var opts = Opts()
|
||||
var args = Array(CommandLine.arguments.dropFirst())
|
||||
while !args.isEmpty {
|
||||
let a = args.removeFirst()
|
||||
func next() -> String { if args.isEmpty { fail("missing value for \(a)") }; return args.removeFirst() }
|
||||
switch a {
|
||||
case "--pack": opts.pack = next()
|
||||
case "--batches": opts.batches = Int(next())!
|
||||
case "--batch-log2": opts.batchLog2 = Int(next())!
|
||||
case "--group": opts.group = Int(next())!
|
||||
case "--warps": opts.warps = Int(next())!
|
||||
default: fail("unknown argument \(a)")
|
||||
}
|
||||
}
|
||||
if opts.pack.isEmpty { fail("--pack <dir> is required") }
|
||||
|
||||
func readText(_ name: String) -> String {
|
||||
guard let s = try? String(contentsOfFile: opts.pack + "/" + name, encoding: .utf8) else { fail("cannot read \(opts.pack)/\(name)") }
|
||||
return s
|
||||
}
|
||||
let programH = readText("program.h")
|
||||
func defineU32(_ name: String) -> UInt32? {
|
||||
let pat = "#define \(name) ([0-9a-fA-Fx]+)"
|
||||
guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
|
||||
let v = String(programH[Range(m.range(at: 1), in: programH)!]).replacingOccurrences(of: "u", with: "")
|
||||
if v.hasPrefix("0x") { return UInt32(v.dropFirst(2), radix: 16) }
|
||||
return UInt32(v)
|
||||
}
|
||||
func defineStr(_ name: String) -> String? {
|
||||
let pat = "#define \(name) \"([^\"]*)\""
|
||||
guard let re = try? NSRegularExpression(pattern: pat), let m = re.firstMatch(in: programH, range: NSRange(programH.startIndex..., in: programH)) else { return nil }
|
||||
return String(programH[Range(m.range(at: 1), in: programH)!])
|
||||
}
|
||||
let datasetLog2 = Int(defineU32("IGNEUM_DATASET_LOG2") ?? 28)
|
||||
let datasetMode = defineU32("IGNEUM_DATASET_MODE") ?? 1
|
||||
if datasetMode != 1 { fail("packbench runs memory-hard packs only") }
|
||||
let cacheLog2 = Int(defineU32("IGNEUM_CACHE_LOG2_WORDS") ?? 26)
|
||||
let cacheSegments = Int(defineU32("IGNEUM_CACHE_SEGMENTS") ?? 65536)
|
||||
let loadsPerHash = Int(defineU32("IGNEUM_LOADS_PER_HASH") ?? 128)
|
||||
let bytesPerHash = Int(defineU32("IGNEUM_BYTES_PER_HASH") ?? UInt32(loadsPerHash * 4))
|
||||
let scratchOps = Int(defineU32("IGNEUM_SCRATCH_OPS") ?? 0)
|
||||
let persistent = (defineU32("IGNEUM_PERSISTENT_WARPS") ?? 0) == 1
|
||||
let scratchWordsPerLane = Int(defineU32("IGNEUM_SCRATCH_WORDS_PER_LANE") ?? 8192)
|
||||
let className = defineStr("IGNEUM_LOAD_CLASS") ?? "v2"
|
||||
let seedString = defineStr("IGNEUM_SEED_STRING") ?? "?"
|
||||
let programId = defineStr("IGNEUM_PROGRAM_ID") ?? ""
|
||||
|
||||
// vectors.json: bases, expected outputs, dataset head / last, cache fingerprint
|
||||
let vj = try! JSONSerialization.jsonObject(with: Data(contentsOf: URL(fileURLWithPath: opts.pack + "/vectors.json"))) as! [String: Any]
|
||||
func hex64(_ s: String) -> UInt64 { return UInt64(s.dropFirst(2), radix: 16)! }
|
||||
func hex32(_ s: String) -> UInt32 { return UInt32(s.dropFirst(2), radix: 16)! }
|
||||
let warpsJ = vj["warps"] as! [[String: Any]]
|
||||
let vecBases = warpsJ.map { UInt32(($0["base_nonce"] as! NSNumber).uint64Value) }
|
||||
let vecOuts = warpsJ.map { ($0["expected"] as! [String]).map(hex64) }
|
||||
let dsHead = (vj["dataset_head"] as! [String]).map(hex32)
|
||||
let dsLastIndex = UInt32((vj["dataset_last_index"] as! NSNumber).uint64Value)
|
||||
let dsLast = hex32(vj["dataset_last"] as! String)
|
||||
let cacheFnvWant = hex64(vj["cache_fnv1a64"] as! String)
|
||||
|
||||
guard let device = MTLCreateSystemDefaultDevice(), let queue = device.makeCommandQueue() else { fail("no Metal device") }
|
||||
let words = 1 << datasetLog2
|
||||
let mask = UInt32(words - 1)
|
||||
let cacheWords = 1 << cacheLog2
|
||||
|
||||
func compile(_ file: String) -> MTLLibrary {
|
||||
do { return try device.makeLibrary(source: readText(file), options: MTLCompileOptions()) } catch { fail("Metal compile of \(file): \(error)") }
|
||||
}
|
||||
let t0 = nowMs()
|
||||
let mhLib = compile("memhard.metal")
|
||||
let progLib = compile("program.metal")
|
||||
guard let fillFn = mhLib.makeFunction(name: "igneum_cache_fill"), let buildFn = mhLib.makeFunction(name: "igneum_build"), let hashFn = progLib.makeFunction(name: "igneum_hash") else { fail("kernel functions missing") }
|
||||
let fillPipe = try! device.makeComputePipelineState(function: fillFn)
|
||||
let buildPipe = try! device.makeComputePipelineState(function: buildFn)
|
||||
let hashPipe = try! device.makeComputePipelineState(function: hashFn)
|
||||
let compileMs = nowMs() - t0
|
||||
if hashPipe.threadExecutionWidth != 32 { print("WARNING: threadExecutionWidth \(hashPipe.threadExecutionWidth), not 32") }
|
||||
|
||||
guard let cache = device.makeBuffer(length: cacheWords * 4, options: .storageModePrivate) else { fail("cache alloc") }
|
||||
guard let dataset = device.makeBuffer(length: words * 4, options: .storageModePrivate) else { fail("dataset alloc") }
|
||||
|
||||
func run(_ body: (MTLComputeCommandEncoder) -> Void) -> (Double, Double) {
|
||||
let cb = queue.makeCommandBuffer()!
|
||||
let enc = cb.makeComputeCommandEncoder()!
|
||||
body(enc)
|
||||
enc.endEncoding()
|
||||
let w0 = nowMs()
|
||||
cb.commit(); cb.waitUntilCompleted()
|
||||
if let e = cb.error { fail("command buffer: \(e)") }
|
||||
return (nowMs() - w0, (cb.gpuEndTime - cb.gpuStartTime) * 1000)
|
||||
}
|
||||
let (cacheWall, cacheGpu) = run { enc in
|
||||
enc.setComputePipelineState(fillPipe); enc.setBuffer(cache, offset: 0, index: 0)
|
||||
enc.dispatchThreadgroups(MTLSize(width: cacheSegments / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
||||
}
|
||||
let items = words / 16
|
||||
let (buildWall, buildGpu) = run { enc in
|
||||
enc.setComputePipelineState(buildPipe); enc.setBuffer(cache, offset: 0, index: 0); enc.setBuffer(dataset, offset: 0, index: 1)
|
||||
enc.dispatchThreadgroups(MTLSize(width: items / 256, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: 256, height: 1, depth: 1))
|
||||
}
|
||||
// cache fingerprint and dataset head/last through a blit to shared memory
|
||||
func blit(_ src: MTLBuffer, _ offset: Int, _ n: Int) -> MTLBuffer {
|
||||
let dst = device.makeBuffer(length: n, options: .storageModeShared)!
|
||||
let cb = queue.makeCommandBuffer()!; let b = cb.makeBlitCommandEncoder()!
|
||||
b.copy(from: src, sourceOffset: offset, to: dst, destinationOffset: 0, size: n); b.endEncoding(); cb.commit(); cb.waitUntilCompleted()
|
||||
return dst
|
||||
}
|
||||
func fnv1a64(_ p: UnsafeRawPointer, _ n: Int) -> UInt64 {
|
||||
var h: UInt64 = 0xcbf29ce484222325
|
||||
let b = p.bindMemory(to: UInt8.self, capacity: n)
|
||||
for i in 0..<n { h ^= UInt64(b[i]); h = h &* 0x100000001b3 }
|
||||
return h
|
||||
}
|
||||
let cacheCopy = blit(cache, 0, cacheWords * 4)
|
||||
let cacheFnv = fnv1a64(cacheCopy.contents(), cacheWords * 4)
|
||||
let cacheOk = cacheFnv == cacheFnvWant
|
||||
let headCopy = blit(dataset, 0, 64)
|
||||
let headPtr = headCopy.contents().bindMemory(to: UInt32.self, capacity: 16)
|
||||
var dsOk = (0..<16).allSatisfy { headPtr[$0] == dsHead[$0] }
|
||||
let lastCopy = blit(dataset, Int(dsLastIndex) * 4, 4)
|
||||
dsOk = dsOk && lastCopy.contents().bindMemory(to: UInt32.self, capacity: 1)[0] == dsLast
|
||||
|
||||
// scratch arena (variant 5)
|
||||
var scratch: MTLBuffer? = nil
|
||||
var salt: UInt32 = 1
|
||||
let warpsN = persistent ? opts.warps : 0
|
||||
if persistent {
|
||||
let bytes = warpsN * 32 * scratchWordsPerLane * 4
|
||||
guard let s = device.makeBuffer(length: bytes, options: .storageModePrivate) else { fail("scratch alloc of \(bytes >> 20) MiB") }
|
||||
scratch = s
|
||||
}
|
||||
// One hash launch: `nonces` outputs from `base`. Persistent: warpsN warps loop over nonces / 32 units.
|
||||
func encodeHash(_ enc: MTLComputeCommandEncoder, out: MTLBuffer, base: UInt32, nonces: Int, group: Int) {
|
||||
enc.setComputePipelineState(hashPipe)
|
||||
enc.setBuffer(dataset, offset: 0, index: 0)
|
||||
enc.setBuffer(out, offset: 0, index: 1)
|
||||
var b = base; enc.setBytes(&b, length: 4, index: 2)
|
||||
if persistent {
|
||||
let units = nonces / 32
|
||||
let nw = min(warpsN, units)
|
||||
if units % nw != 0 { fail("nonces \(nonces) is not a multiple of 32 x \(nw) warps") }
|
||||
enc.setBuffer(scratch!, offset: 0, index: 3)
|
||||
var g = UInt32(units); enc.setBytes(&g, length: 4, index: 4)
|
||||
var s = salt; enc.setBytes(&s, length: 4, index: 5)
|
||||
salt = salt &+ UInt32(units)
|
||||
let threads = nw * 32
|
||||
let tg = min(group, threads)
|
||||
enc.dispatchThreadgroups(MTLSize(width: threads / tg, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1))
|
||||
} else {
|
||||
let tg = min(group, nonces)
|
||||
enc.dispatchThreadgroups(MTLSize(width: nonces / tg, height: 1, depth: 1), threadsPerThreadgroup: MTLSize(width: tg, height: 1, depth: 1))
|
||||
}
|
||||
}
|
||||
// Vectors, standalone (one unit per launch)
|
||||
var vecPass = 0
|
||||
let vecOut = device.makeBuffer(length: 32 * 8, options: .storageModeShared)!
|
||||
for (i, base) in vecBases.enumerated() {
|
||||
_ = run { enc in encodeHash(enc, out: vecOut, base: base, nonces: 32, group: 32) }
|
||||
let p = vecOut.contents().bindMemory(to: UInt64.self, capacity: 32)
|
||||
var ok = true
|
||||
for l in 0..<32 where p[l] != vecOuts[i][l] { ok = false; print("vector warp base \(base) lane \(l): GPU \(String(format: "%016llx", p[l])) expected \(String(format: "%016llx", vecOuts[i][l]))"); break }
|
||||
if ok { vecPass += 1 }
|
||||
}
|
||||
// Batch at base 0: fingerprint and the vectors inside the batch
|
||||
var nonces = 1 << opts.batchLog2
|
||||
if persistent { let unit = 32 * min(warpsN, nonces / 32); nonces = (nonces / unit) * unit }
|
||||
let out = device.makeBuffer(length: nonces * 8, options: .storageModeShared)!
|
||||
let (warmWall, warmGpu) = run { enc in encodeHash(enc, out: out, base: 0, nonces: nonces, group: opts.group) }
|
||||
let outPtr = out.contents().bindMemory(to: UInt64.self, capacity: nonces)
|
||||
var batchVecPass = 0, batchVecN = 0
|
||||
for (i, base) in vecBases.enumerated() where Int(base) + 32 <= nonces {
|
||||
batchVecN += 1
|
||||
if (0..<32).allSatisfy({ outPtr[Int(base) + $0] == vecOuts[i][$0] }) { batchVecPass += 1 }
|
||||
}
|
||||
let fingerprint = fnv1a64(out.contents(), nonces * 8)
|
||||
// Timed batches
|
||||
var wallSum = 0.0, gpuSum = 0.0
|
||||
for b in 0..<opts.batches {
|
||||
let (w, g) = run { enc in encodeHash(enc, out: out, base: UInt32(truncatingIfNeeded: (b + 1) * nonces), nonces: nonces, group: opts.group) }
|
||||
wallSum += w; gpuSum += g
|
||||
}
|
||||
let hashes = Double(nonces) * Double(opts.batches)
|
||||
let mhsWall = hashes / wallSum / 1e3, mhsGpu = hashes / gpuSum / 1e3
|
||||
let packName = (opts.pack as NSString).lastPathComponent
|
||||
print("pack \(packName) seed \"\(seedString)\" id \(programId) class \(className): loads/hash \(loadsPerHash), dataset bytes/hash \(bytesPerHash), scratch ops/hash \(scratchOps * 8)")
|
||||
print("device \(device.name); compile \(String(format: "%.0f", compileMs)) ms; cache fill \(String(format: "%.1f", cacheGpu)) ms GPU (\(String(format: "%.1f", cacheWall)) wall); dataset build \(String(format: "%.1f", buildGpu)) ms GPU (\(String(format: "%.1f", buildWall)) wall)")
|
||||
print("cache FNV-1a 64 \(String(format: "%016llx", cacheFnv)) \(cacheOk ? "PASS" : "FAIL"); dataset head and last \(dsOk ? "PASS" : "FAIL"); vectors standalone \(vecPass)/\(vecBases.count), in batch \(batchVecPass)/\(batchVecN)")
|
||||
print("warm-up batch \(nonces) hashes: \(String(format: "%.1f", warmGpu)) ms GPU, \(String(format: "%.1f", warmWall)) ms wall")
|
||||
let overall = cacheOk && dsOk && vecPass == vecBases.count && batchVecPass == batchVecN
|
||||
print("RESULT pack=\(packName) class=\(className) device=\(device.name.replacingOccurrences(of: " ", with: "_")) group=\(opts.group) warps=\(warpsN) arena_mib=\(persistent ? warpsN : 0) nonces=\(nonces) batches=\(opts.batches) vectors=\(vecPass)/\(vecBases.count) batch_vectors=\(batchVecPass)/\(batchVecN) cache=\(cacheOk ? "PASS" : "FAIL") dataset=\(dsOk ? "PASS" : "FAIL") fingerprint=\(String(format: "%016llx", fingerprint)) mhs_gpu=\(String(format: "%.3f", mhsGpu)) mhs_wall=\(String(format: "%.3f", mhsWall)) loads=\(loadsPerHash) bytes=\(bytesPerHash) scratch_ops=\(scratchOps * 8) overall=\(overall ? "PASS" : "FAIL")")
|
||||
exit(overall ? 0 : 1)
|
||||
|
|
@ -41,7 +41,12 @@
|
|||
#endif
|
||||
|
||||
// Kernels from kernel.cl (compiled as a separate C++ translation unit with the same defines).
|
||||
#ifdef IGNEUM_PERSISTENT_WARPS
|
||||
// Variant 5 of the read-width experiment: persistent warps with a 1 MiB scratch each (program.h says so).
|
||||
void igneum_hash(const uint* ds, ulong* out, uint baseNonce, uint mask, uint* scratch, uint groups, uint salt);
|
||||
#else
|
||||
void igneum_hash(const uint* ds, ulong* out, uint baseNonce, uint mask);
|
||||
#endif
|
||||
#if IGNEUM_DATASET_MODE == 1
|
||||
void igneum_cache_fill(uint* cache, uint nSegments);
|
||||
void igneum_build(uint* ds, const uint* cache, uint nItems);
|
||||
|
|
@ -280,14 +285,39 @@ int main(int argc, char** argv) {
|
|||
// Vectors standalone: one work-group of IGNEUM_GROUP items per base nonce (the first 32 are the vector warp).
|
||||
const uint32_t nonces = 1u << batchLog2;
|
||||
std::vector<ulong> out(nonces);
|
||||
#ifdef IGNEUM_PERSISTENT_WARPS
|
||||
// Variant 5: EMU_WARPS persistent warps (one per work-group of 32), each with IGNEUM_SCRATCH_WORDS_PER_LANE x 32 words.
|
||||
const unsigned emuWarps = 8;
|
||||
std::vector<uint> scratchArena((size_t)emuWarps * 32u * IGNEUM_SCRATCH_WORDS_PER_LANE, 0u);
|
||||
uint salt = 1u;
|
||||
if (IGNEUM_GROUP != 32) { std::printf("variant 5 needs IGNEUM_GROUP 32 (one warp per work-group)\n"); return 2; }
|
||||
std::printf("variant 5: %u persistent warps, scratch arena %u MiB, lazy tagged fill\n", emuWarps, (unsigned)((scratchArena.size() * 4u) >> 20));
|
||||
auto launchHash = [&](size_t units, uint base) {
|
||||
size_t nw = units < emuWarps ? units : emuWarps;
|
||||
emu_launch(igneum_hash, nw * 32u, 32u, (const uint*)ds.data(), out.data(), base, mask, scratchArena.data(), (uint)units, salt);
|
||||
salt += (uint)units;
|
||||
};
|
||||
#else
|
||||
auto launchHash = [&](size_t nonceCount, uint base) {
|
||||
emu_launch(igneum_hash, nonceCount, (unsigned)IGNEUM_GROUP, (const uint*)ds.data(), out.data(), base, mask);
|
||||
};
|
||||
#endif
|
||||
for (int w = 0; w < IGNEUM_VEC_WARPS; ++w) {
|
||||
emu_launch(igneum_hash, (size_t)IGNEUM_GROUP, (unsigned)IGNEUM_GROUP, (const uint*)ds.data(), out.data(), (uint)IGNEUM_VEC_BASE[w], mask);
|
||||
#ifdef IGNEUM_PERSISTENT_WARPS
|
||||
launchHash(1, (uint)IGNEUM_VEC_BASE[w]);
|
||||
#else
|
||||
launchHash((size_t)IGNEUM_GROUP, (uint)IGNEUM_VEC_BASE[w]);
|
||||
#endif
|
||||
char how[96];
|
||||
std::snprintf(how, sizeof(how), "standalone, work-group %d, sub-group width %u", IGNEUM_GROUP, gSubGroupWidth);
|
||||
overall = compareWarp((const uint64_t*)out.data(), IGNEUM_VEC_OUT[w], IGNEUM_VEC_BASE[w], how) && overall;
|
||||
}
|
||||
// In batch: every vector warp that fits in 2^batchLog2 nonces.
|
||||
emu_launch(igneum_hash, (size_t)nonces, (unsigned)IGNEUM_GROUP, (const uint*)ds.data(), out.data(), 0u, mask);
|
||||
#ifdef IGNEUM_PERSISTENT_WARPS
|
||||
launchHash(nonces / 32u, 0u);
|
||||
#else
|
||||
launchHash((size_t)nonces, 0u);
|
||||
#endif
|
||||
for (int w = 0; w < IGNEUM_VEC_WARPS; ++w) {
|
||||
if ((uint64_t)IGNEUM_VEC_BASE[w] + 32ull > nonces) { std::printf("verify warp base %u in batch: skipped (batch has %u nonces)\n", IGNEUM_VEC_BASE[w], nonces); continue; }
|
||||
char how[96];
|
||||
|
|
|
|||
|
|
@ -207,6 +207,9 @@ typedef struct {
|
|||
int memprobe; // --memprobe: dependent-load latency and throughput, independent-load throughput and an ALU
|
||||
// chain on the chosen device, no pack needed (5 October 2026, the 9070 XT on the eGPU)
|
||||
int probeMib; // --probe-mib N: --memprobe at that one buffer size only (default 0 = 4, 64 and 1024 MiB)
|
||||
int benchPack; // --bench-pack: with --pack D, build and self-test the pack at run time (as --serve does) and time
|
||||
// igneum_hash_bound with the pack's seed words as init words; read-width experiment, 5 October 2026
|
||||
int warps; // --warps N: persistent warps for a variant-5 pack (IGNEUM_PERSISTENT_WARPS); 0 = 2048
|
||||
} Options;
|
||||
|
||||
static int packMib(void) { return (int)(((1ull << IGNEUM_DATASET_LOG2) * 4ull) >> 20); }
|
||||
|
|
@ -236,6 +239,8 @@ static void usage(void) {
|
|||
" pack this exe was built against; it is self-tested against its vectors.h first (the one-click worker)\n"
|
||||
" --readback M with --serve: select (default) reads back only the hits and 34 sentinel words of each dispatch through a\n"
|
||||
" GPU-side pass; full reads back every output (8 bytes per nonce). IGNEUM_READBACK=full does the same.\n"
|
||||
" --bench-pack with --pack D: read the pack at run time, build and self-test it, time its bound kernel (one exe, any pack)\n"
|
||||
" --warps N persistent warps for a variant-5 pack (a 1 MiB scratch each; default 2048; the batch rounds to 32 x N)\n"
|
||||
" --memprobe no pack: dependent random loads (latency and throughput against lanes in flight), independent random\n"
|
||||
" loads and an ALU chain on the chosen device, at 4, 64 and 1024 MiB (--probe-mib N for one size)\n", packMib(), IGNEUM_KERNEL_PATH);
|
||||
}
|
||||
|
|
@ -248,7 +253,7 @@ static Options parseArgs(int argc, char** argv) {
|
|||
int i;
|
||||
o.datasetMib = 1024; o.batchLog2 = 24; o.batches = 5; o.groupWarps = 1; o.sweep = 0; o.device = -1;
|
||||
o.exchange = 0; o.list = 0; o.timeWall = -1; o.kernelPath = IGNEUM_KERNEL_PATH; o.extraOpts = ""; o.serve = 0; o.noPrepare = 0; o.kernelGiven = 0; o.vendor = NULL; o.packDir = NULL;
|
||||
o.readback = (getenv("IGNEUM_READBACK") && strcmp(getenv("IGNEUM_READBACK"), "full") == 0) ? 1 : 0; o.memprobe = 0; o.probeMib = 0;
|
||||
o.readback = (getenv("IGNEUM_READBACK") && strcmp(getenv("IGNEUM_READBACK"), "full") == 0) ? 1 : 0; o.memprobe = 0; o.probeMib = 0; o.benchPack = 0; o.warps = 0;
|
||||
for (i = 1; i < argc; ++i) {
|
||||
const char* a = argv[i];
|
||||
int needs = (strcmp(a, "--dataset-mib") == 0 || strcmp(a, "--batch-log2") == 0 || strcmp(a, "--batches") == 0 ||
|
||||
|
|
@ -274,6 +279,8 @@ static Options parseArgs(int argc, char** argv) {
|
|||
else { printf("--readback must be select or full\n"); exit(2); }
|
||||
}
|
||||
else if (strcmp(a, "--memprobe") == 0) o.memprobe = 1;
|
||||
else if (strcmp(a, "--bench-pack") == 0) o.benchPack = 1;
|
||||
else if (strcmp(a, "--warps") == 0) { if (i + 1 >= argc) { usage(); exit(2); } o.warps = atoi(argv[++i]); }
|
||||
else if (strcmp(a, "--probe-mib") == 0) { if (i + 1 >= argc) { usage(); exit(2); } o.probeMib = atoi(argv[++i]); }
|
||||
else if (strcmp(a, "--build-opts") == 0) o.extraOpts = argv[++i];
|
||||
else if (strcmp(a, "--time") == 0) {
|
||||
|
|
@ -1013,6 +1020,19 @@ static int unhexBuf(const char* s, uint8_t* out, size_t cap, size_t* len) {
|
|||
* from the compiled-in program.h, so one prebuilt exe serves every pack. The compiled-in values are the defaults. */
|
||||
static PfPack gPack;
|
||||
static int gGeneric = 0;
|
||||
/* Variant 5 of the read-width experiment (5 October 2026): the scratch arena of a persistent-warp pack, its warp count
|
||||
* and the running tag salt; set by --bench-pack before the self-test. Serve mode does not support these packs. */
|
||||
static cl_mem gScratch = NULL;
|
||||
static cl_uint gScratchWarps = 0;
|
||||
static cl_uint gSalt = 1;
|
||||
/* Sets the three extra arguments of a variant-5 kernel (after the five of igneum_hash_bound) for `units` units. */
|
||||
static cl_int setScratchArgs(cl_kernel k, cl_uint firstArg, cl_uint units) {
|
||||
cl_int e = clSetKernelArg(k, firstArg, sizeof(cl_mem), &gScratch);
|
||||
if (e == CL_SUCCESS) e = clSetKernelArg(k, firstArg + 1, sizeof(cl_uint), &units);
|
||||
if (e == CL_SUCCESS) e = clSetKernelArg(k, firstArg + 2, sizeof(cl_uint), &gSalt);
|
||||
gSalt += units;
|
||||
return e;
|
||||
}
|
||||
static uint32_t gServeWords = 1u << IGNEUM_DATASET_LOG2;
|
||||
#if IGNEUM_DATASET_MODE == 1
|
||||
static uint32_t gServeCacheWords = 1u << IGNEUM_CACHE_LOG2_WORDS;
|
||||
|
|
@ -1102,6 +1122,10 @@ static int pairSelfTest(Device* dv, const DeviceInfo* di, cl_command_queue q, Se
|
|||
out = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, g * sizeof(uint64_t), NULL, &e);
|
||||
if (e == CL_SUCCESS) { ++gMemCreated; init = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY | CL_MEM_COPY_HOST_PTR, 32, pk.seedw, &e); }
|
||||
if (e == CL_SUCCESS) ++gMemCreated;
|
||||
if (pk.persistent) {
|
||||
if (!gScratch) { snprintf(err, errCap, "self-test: a variant-5 pack (persistent warps) needs --bench-pack (serve mode does not carry a scratch)"); return 0; }
|
||||
g = 32; local = 32; /* one persistent warp runs the one unit */
|
||||
}
|
||||
for (w = 0; w < pk.vecWarps && e == CL_SUCCESS; ++w) {
|
||||
cl_uint base = pk.vecBase[w], mask = words - 1u;
|
||||
e = clSetKernelArg(p->kHashBound, 0, sizeof(cl_mem), &p->ds);
|
||||
|
|
@ -1109,6 +1133,7 @@ static int pairSelfTest(Device* dv, const DeviceInfo* di, cl_command_queue q, Se
|
|||
if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 2, sizeof(cl_uint), &base);
|
||||
if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 3, sizeof(cl_uint), &mask);
|
||||
if (e == CL_SUCCESS) e = clSetKernelArg(p->kHashBound, 4, sizeof(cl_mem), &init);
|
||||
if (e == CL_SUCCESS && pk.persistent) e = setScratchArgs(p->kHashBound, 5, 1u);
|
||||
if (e == CL_SUCCESS) e = clEnqueueNDRangeKernel(q, p->kHashBound, 1, NULL, &g, &local, 0, NULL, NULL);
|
||||
if (e == CL_SUCCESS) e = clEnqueueReadBuffer(q, out, CL_TRUE, 0, 32 * sizeof(uint64_t), &vec[w * 32], 0, NULL, NULL);
|
||||
}
|
||||
|
|
@ -1209,6 +1234,92 @@ static int startPrepareThread(PrepareTask* t) { pthread_t th; if (pthread_create
|
|||
#endif
|
||||
#endif
|
||||
|
||||
/* --bench-pack (read-width experiment, 5 October 2026): the pack in --pack is built and self-tested exactly as the
|
||||
* first pair of --serve (pairBuffers: cache, dataset, cache FNV, dataset words, the vector warps through
|
||||
* igneum_hash_bound with the pack's seed words), then the bound kernel is timed over --batches dispatches of
|
||||
* 2^--batch-log2 nonces with device event time, and the 2^B outputs at base nonce 0 are fingerprinted (FNV-1a 64) so
|
||||
* the same pack can be compared bit for bit across vendors. A variant-5 pack is launched as --warps persistent warps
|
||||
* with a 1 MiB scratch each. One line per run starts with RESULT. */
|
||||
static int runBenchPack(Device* dv, const DeviceInfo* di, const Options* o) {
|
||||
#if IGNEUM_DATASET_MODE != 1
|
||||
(void)dv; (void)di; (void)o;
|
||||
printf("FAIL: --bench-pack needs a memory-hard placeholder pack\n");
|
||||
return 2;
|
||||
#else
|
||||
const uint32_t words = gServeWords, mask = words - 1u;
|
||||
uint32_t nonces = 1u << o->batchLog2;
|
||||
size_t groupSize = 32 * (size_t)o->groupWarps, g;
|
||||
cl_int err = 0;
|
||||
cl_mem dOut, dInit;
|
||||
uint64_t* hOut;
|
||||
ServePair* cur;
|
||||
char perr[512], devName[256];
|
||||
double t0 = wallMs(), sum = 0, warmMs;
|
||||
uint64_t fp;
|
||||
int b, k;
|
||||
cl_uint warps = (cl_uint)(o->warps > 0 ? o->warps : 2048), units = nonces / 32u;
|
||||
if (!dv->kHashBound) { printf("FAIL: the kernel source has no igneum_hash_bound\n"); return 2; }
|
||||
if (gPack.persistent) {
|
||||
size_t arena;
|
||||
if (o->groupWarps != 1) { printf("FAIL: a variant-5 pack needs --group-warps 1 (one warp per work-group: the loop trip count must be uniform)\n"); return 2; }
|
||||
while (warps > 1 && units % warps != 0) warps >>= 1;
|
||||
arena = (size_t)warps * 32u * (size_t)gPack.scratchWordsPerLane * 4u;
|
||||
if ((uint64_t)arena > di->maxAlloc) { printf("FAIL: scratch arena %llu MiB exceeds the device's max alloc %llu MiB; lower --warps\n", (unsigned long long)(arena >> 20), (unsigned long long)(di->maxAlloc >> 20)); return 2; }
|
||||
gScratch = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, arena, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer scratch");
|
||||
gScratchWarps = warps;
|
||||
printf("variant 5: %u persistent warps (%u x 32 work-items, work-group 32), scratch arena %llu MiB, %u units per dispatch, lazy tagged fill\n", warps, warps, (unsigned long long)(arena >> 20), units);
|
||||
}
|
||||
cur = (ServePair*)calloc(1, sizeof(ServePair));
|
||||
cur->kHashBound = dv->kHashBound; cur->kCacheFill = dv->kCacheFill; cur->kBuild = dv->kBuild; cur->prog = dv->prog;
|
||||
dv->kHashBound = dv->kCacheFill = dv->kBuild = NULL; dv->prog = NULL;
|
||||
memcpy(cur->sw, gPack.seedw, 32); memcpy(cur->kw, gPack.keyw, 32);
|
||||
if (!pairBuffers(dv, di, dv->q, cur, words, gServeCacheWords, gServeSegments, o->packDir, perr, sizeof(perr))) { printf("FAIL: pack %s: %s\n", o->packDir, perr); return 1; }
|
||||
printf("pack %s: cache %.0f dataset %.0f check %.0f ms (%.0f ms in all); %s\n", o->packDir, cur->cacheMs, cur->datasetMs, cur->checkMs, wallMs() - t0, cur->check);
|
||||
printKernelInfo(di, cur->kHashBound, "igneum_hash_bound", (int)groupSize, "");
|
||||
dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, (size_t)nonces * sizeof(uint64_t), NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer out");
|
||||
dInit = clCreateBuffer(dv->ctx, CL_MEM_READ_ONLY | CL_MEM_COPY_HOST_PTR, 32, cur->sw, &err); CL_CHECK_ERR(err, "clCreateBuffer init words");
|
||||
hOut = (uint64_t*)malloc((size_t)nonces * sizeof(uint64_t));
|
||||
strncpy(devName, di->name, 255); devName[255] = 0;
|
||||
for (k = 0; devName[k]; ++k) if (devName[k] == ' ') devName[k] = '_';
|
||||
g = gPack.persistent ? (size_t)warps * 32u : (size_t)nonces;
|
||||
for (b = -1; b < o->batches; ++b) {
|
||||
cl_uint base = (cl_uint)((uint32_t)(b + 1) * nonces);
|
||||
cl_event ev = NULL;
|
||||
double ms;
|
||||
CL_CHECK(clSetKernelArg(cur->kHashBound, 0, sizeof(cl_mem), &cur->ds));
|
||||
CL_CHECK(clSetKernelArg(cur->kHashBound, 1, sizeof(cl_mem), &dOut));
|
||||
CL_CHECK(clSetKernelArg(cur->kHashBound, 2, sizeof(cl_uint), &base));
|
||||
CL_CHECK(clSetKernelArg(cur->kHashBound, 3, sizeof(cl_uint), &mask));
|
||||
CL_CHECK(clSetKernelArg(cur->kHashBound, 4, sizeof(cl_mem), &dInit));
|
||||
if (gPack.persistent) CL_CHECK(setScratchArgs(cur->kHashBound, 5, units));
|
||||
{
|
||||
double w0 = wallMs();
|
||||
CL_CHECK(clEnqueueNDRangeKernel(dv->q, cur->kHashBound, 1, NULL, &g, &groupSize, 0, NULL, &ev));
|
||||
CL_CHECK(clWaitForEvents(1, &ev));
|
||||
ms = o->timeWall ? wallMs() - w0 : eventMs(ev);
|
||||
if (ms < 0) ms = wallMs() - w0;
|
||||
clReleaseEvent(ev);
|
||||
}
|
||||
if (b < 0) {
|
||||
warmMs = ms;
|
||||
CL_CHECK(clEnqueueReadBuffer(dv->q, dOut, CL_TRUE, 0, (size_t)nonces * sizeof(uint64_t), hOut, 0, NULL, NULL));
|
||||
fp = pf_fnv1a64((const uint32_t*)hOut, (size_t)nonces * 8u);
|
||||
} else sum += ms;
|
||||
}
|
||||
printf("warm-up dispatch (base 0): %.2f ms; %d timed dispatches of 2^%d nonces: mean %.2f ms\n", warmMs, o->batches, o->batchLog2, sum / o->batches);
|
||||
printf("RESULT pack=%s class=%s device=%s platform=%s group=%d warps=%u arena_mib=%llu nonces=%u batches=%d check=%s fingerprint=%016llx mhs=%.3f loads=%u bytes=%u scratch_ops=%u time=%s\n",
|
||||
o->packDir, gPack.loadClass, devName, strcmp(di->platformName, "Apple") == 0 ? "Apple" : "other", (int)groupSize, gPack.persistent ? warps : 0u,
|
||||
gPack.persistent ? (unsigned long long)(((size_t)warps * 32u * gPack.scratchWordsPerLane * 4u) >> 20) : 0ull, nonces, o->batches,
|
||||
cur->checked ? "PASS" : "skipped", (unsigned long long)fp, (double)nonces * (double)o->batches / (sum / 1000.0) / 1e6,
|
||||
gPack.loadsPerHash, gPack.bytesPerHash, gPack.scratchOps * 8u, o->timeWall ? "wall" : "event");
|
||||
free(hOut);
|
||||
clReleaseMemObject(dOut); clReleaseMemObject(dInit);
|
||||
if (gScratch) clReleaseMemObject(gScratch);
|
||||
releasePair(cur);
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
static int runServe(Device* dv, const DeviceInfo* di, const Options* o) {
|
||||
#if IGNEUM_DATASET_MODE != 1
|
||||
(void)dv; (void)di; (void)o;
|
||||
|
|
@ -1612,6 +1723,11 @@ static const char* PROBE_SRC =
|
|||
" }\n"
|
||||
" out[g] = x0 ^ x1 ^ x2 ^ x3 ^ x4 ^ x5 ^ x6 ^ x7;\n"
|
||||
"}\n"
|
||||
"__kernel void probe_line16(__global const uint4* ds, uint vecMask, uint steps, uint seed, __global uint* out) {\n"
|
||||
" uint x = pm_mix((uint)get_global_id(0) ^ seed);\n"
|
||||
" for (uint s = 0u; s < steps; ++s) { uint4 a = ds[x & vecMask]; x = (a.x ^ a.y ^ a.z ^ a.w) ^ (x * 0x9E3779B1u + s); }\n"
|
||||
" out[get_global_id(0)] = x;\n"
|
||||
"}\n"
|
||||
"__kernel void probe_line(__global const uint4* ds, uint lineMask, uint steps, uint seed, __global uint* out) {\n"
|
||||
" uint x = pm_mix((uint)get_global_id(0) ^ seed);\n"
|
||||
" for (uint s = 0u; s < steps; ++s) {\n"
|
||||
|
|
@ -1656,7 +1772,7 @@ static double probeLaunch(Device* dv, const Options* o, cl_kernel k, size_t glob
|
|||
static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) {
|
||||
cl_int err = 0;
|
||||
cl_program prog;
|
||||
cl_kernel kFill, kChase, kIndep, kAlu, kLine, kStream;
|
||||
cl_kernel kFill, kChase, kIndep, kAlu, kLine, kStream, kLine16;
|
||||
size_t srcLen = strlen(PROBE_SRC);
|
||||
int sizes[3] = { 4, 64, 1024 }, nSizes = 3, si;
|
||||
size_t lanesList[8] = { 256, 1024, 1u << 12, 1u << 14, 1u << 16, 1u << 18, 1u << 20, 1u << 22 };
|
||||
|
|
@ -1682,6 +1798,7 @@ static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
kIndep = clCreateKernel(prog, "probe_indep", &err); CL_CHECK_ERR(err, "probe_indep");
|
||||
kAlu = clCreateKernel(prog, "probe_alu", &err); CL_CHECK_ERR(err, "probe_alu");
|
||||
kLine = clCreateKernel(prog, "probe_line", &err); CL_CHECK_ERR(err, "probe_line");
|
||||
kLine16 = clCreateKernel(prog, "probe_line16", &err); CL_CHECK_ERR(err, "probe_line16");
|
||||
kStream = clCreateKernel(prog, "probe_stream", &err); CL_CHECK_ERR(err, "probe_stream");
|
||||
dOut = clCreateBuffer(dv->ctx, CL_MEM_READ_WRITE, maxLanes * 4u, NULL, &err); CL_CHECK_ERR(err, "clCreateBuffer probe out");
|
||||
printf("memprobe on [%s] %s, driver %s, %u compute units, %u MHz, %s time\n", di->platformName, di->name, di->driver, di->computeUnits, di->clockMHz, o->timeWall ? "wall" : "device event");
|
||||
|
|
@ -1737,6 +1854,25 @@ static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
fflush(stdout);
|
||||
}
|
||||
}
|
||||
{
|
||||
/* Random 16-byte reads (one uint4) in a dependent chain: the W = 16 width of the read-width experiment. */
|
||||
size_t local = di->maxWorkGroup < 256 ? di->maxWorkGroup : 256;
|
||||
size_t lanes;
|
||||
cl_uint vecMask = (words / 4u) - 1u;
|
||||
for (lanes = 1u << 14; lanes <= maxLanes; lanes <<= 2) {
|
||||
cl_uint seed = 0x2718281u;
|
||||
double ms;
|
||||
CL_CHECK(clSetKernelArg(kLine16, 0, sizeof(cl_mem), &dDs));
|
||||
CL_CHECK(clSetKernelArg(kLine16, 1, sizeof(cl_uint), &vecMask));
|
||||
CL_CHECK(clSetKernelArg(kLine16, 2, sizeof(cl_uint), &STEPS));
|
||||
CL_CHECK(clSetKernelArg(kLine16, 3, sizeof(cl_uint), &seed));
|
||||
CL_CHECK(clSetKernelArg(kLine16, 4, sizeof(cl_mem), &dOut));
|
||||
ms = probeLaunch(dv, o, kLine16, lanes, local, 3, 3, seed);
|
||||
printf("| line 16 B | %d | %llu | %llu | %u | %.3f | %.3f G reads/s | %.1f GB/s in 16 B reads |\n", mib, (unsigned long long)local, (unsigned long long)lanes, STEPS, ms,
|
||||
(double)lanes * (double)STEPS / (ms / 1000.0) / 1e9, (double)lanes * (double)STEPS * 16.0 / (ms / 1000.0) / 1e9);
|
||||
fflush(stdout);
|
||||
}
|
||||
}
|
||||
{
|
||||
/* Random 64-byte lines (16 words, four uint4 loads) in a dependent chain: lines per second against the
|
||||
* 4-byte chase above says what one random 4-byte read costs the memory system. If the two rates are
|
||||
|
|
@ -1792,7 +1928,7 @@ static int runMemprobe(Device* dv, const DeviceInfo* di, const Options* o) {
|
|||
(double)lanes * (double)ALU_STEPS / (ms / 1000.0) / 1e9 / (double)(di->computeUnits ? di->computeUnits : 1));
|
||||
}
|
||||
clReleaseMemObject(dOut);
|
||||
clReleaseKernel(kFill); clReleaseKernel(kChase); clReleaseKernel(kIndep); clReleaseKernel(kAlu); clReleaseKernel(kLine); clReleaseKernel(kStream);
|
||||
clReleaseKernel(kFill); clReleaseKernel(kChase); clReleaseKernel(kIndep); clReleaseKernel(kAlu); clReleaseKernel(kLine); clReleaseKernel(kStream); clReleaseKernel(kLine16);
|
||||
clReleaseProgram(prog);
|
||||
printf("memprobe: done\n");
|
||||
return 0;
|
||||
|
|
@ -1821,7 +1957,7 @@ int main(int argc, char** argv) {
|
|||
static char boundPath[1200];
|
||||
char perr[512];
|
||||
size_t n = strlen(o.packDir);
|
||||
if (!o.serve) { printf("FAIL: --pack goes with --serve (the bench runs the compiled-in pack)\n"); return 2; }
|
||||
if (!o.serve && !o.benchPack) { printf("FAIL: --pack goes with --serve or --bench-pack (the plain bench runs the compiled-in pack)\n"); return 2; }
|
||||
if (n > 1 && (o.packDir[n - 1] == '/' || o.packDir[n - 1] == '\\')) ((char*)o.packDir)[n - 1] = 0;
|
||||
if (!pf_load(o.packDir, &gPack, perr, sizeof(perr))) { printf("error 0 pack %s: %s\n", o.packDir, perr); fflush(stdout); return 2; }
|
||||
gGeneric = 1;
|
||||
|
|
@ -1877,6 +2013,7 @@ int main(int argc, char** argv) {
|
|||
clReleaseContext(dv.ctx);
|
||||
return rc;
|
||||
}
|
||||
if (o.benchPack && !o.packDir) { printf("FAIL: --bench-pack needs --pack <dir>\n"); return 2; }
|
||||
if (o.serve && !o.kernelGiven) {
|
||||
/* The bound kernel lives next to the compiled-in kernel.cl as kernel_bound.cl (packs from igneum-pow or igneum-miner export-pack). */
|
||||
static char boundPath[1024];
|
||||
|
|
@ -1894,6 +2031,7 @@ int main(int argc, char** argv) {
|
|||
printf("build options: %s\n", dv.buildOptions);
|
||||
printf("exchange: %s\n", dv.exchangeNote);
|
||||
if (o.serve) return runServe(&dv, di, &o);
|
||||
if (o.benchPack) return runBenchPack(&dv, di, &o);
|
||||
printKernelInfo(di, dv.kHash, "igneum_hash", dv.groupSize, "");
|
||||
printf("program: %d instructions x %d iterations, loads/hash %d, op mix %s\n",
|
||||
IGNEUM_INSTR_COUNT, IGNEUM_ITERATIONS, IGNEUM_LOADS_PER_HASH, IGNEUM_OP_MIX);
|
||||
|
|
|
|||
Loading…
Reference in a new issue