igneum/igneum-pow/src/emit.rs
igneum-labs fdcab858e3 Lottery hash: generator version 2 (16 load slots, fresh sources, acceptance rule), every vector re-cut, packs regenerated, three workers re-checked, 20,000-program census
igneum-pow 0.2.0: generator v2 draws exactly 16 load slots from instructions 1..63, a
load's source from the registers written earlier and not read by a load since, the other
48 ops from the ten non-load weights; accept.rs is spec 01 section 1.4.6 (static: no
stale load source, every register injected; dynamic: 64 units on the seed-keyed
closed-form dataset, no constant bit, no lane-constant site, under 164 saturated, bias
within 136 of 1024, distinct addresses above 245,760); a rejected candidate is replaced
by the next attempt of the seed (seed || k_le32), 32 a consensus fault. Packs carry the
generator version, attempt and program id. Version 1 kept as generate_v1 for the census.

Packs: igneum-genesis, igneum-hourly, igneum-genesis-mh regenerated by igneum-pow export;
new igneum-devnet-v4-epoch0 (devnet genesis hash, day bytes 20730). Checks: Rust 39 of
39 tests; Metal natively via the Swift port (export cross-check 3 of 3 warps, identical
programs and vectors on five seeds incl. three with attempt 1, fuzz 2,000 of 2,000);
CUDA emu 4 of 4 packs; OpenCL emu 2 packs x 2 configurations; Apple OpenCL 4 of 4 packs
at 27.9 Mhash/s. Census 20,000: 5.225 percent rejected, accepted distinct mean 127.887.

Spec 01 0.2 (1.4.2, 1.4.3, 1.4.6, 1.11, 1.15, 1.16, 1.17), igneum-pow README, the CUDA,
OpenCL and Metal test notes, bench-log entry, ledger M5 and M6 Fixed.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-04 07:52:40 +00:00

1215 lines
64 KiB
Rust

//! Kernel source emitters. Since 4 October 2026 (generator version 2) this crate is the source of every pack in
//! `proto-cuda/packs/`; the pack tests diff the emitters against the checked-in files. Each function started as a
//! byte-for-byte twin of its namesake in `proto-metal/main.swift` (`generateMSL`, `memhardMSL`, `emitMemhardCore`,
//! `generateCUDA`, `generateOpenCL`, `generateProgramHeader`, `generateMemhardHeader`, `generateVectorsHeader`,
//! `generateProgramJSON`, `generateVectorsJSON`); the kernel text is unchanged by version 2, and `program.json` and
//! `program.h` carry the generator version, the attempt and the program id so no version 1 pack can be mistaken
//! for a current one.
//!
//! One deliberate difference from the Swift: `program_json` writes the cache line mask inside the `"item"` string
//! as a bare `0x003fffff`. The Swift writes it quoted (`jhex`), which is not valid JSON.
use crate::generator::{Op, Program, GENERATOR_VERSION, INSTR_COUNT, ITERATIONS, LOAD_SLOTS};
use crate::memhard::{
MixParams, CACHE_LINES_PER_SEGMENT, CACHE_LINE_MASK, CACHE_LOG2_WORDS, CACHE_SEGMENTS, CACHE_SEGMENT_LOG2_LINES,
CACHE_TAG, CACHE_WORDS, CHACHA_ROUNDS, CHACHA_SIGMA, ITEM_ROUNDS,
};
use crate::seed::SplitMix64;
use crate::verify::{DatasetMode, DatasetSource, Epoch};
pub fn hex(v: u32) -> String {
format!("0x{v:08x}u")
}
pub fn hex64(v: u64) -> String {
format!("0x{v:016x}ull")
}
fn jhex(v: u32) -> String {
format!("\"0x{v:08x}\"")
}
fn jhex64(v: u64) -> String {
format!("\"0x{v:016x}\"")
}
/// JSON string with the three escapes the Swift applies (quote, backslash, newline).
fn jstr(s: &str) -> String {
let mut o = String::with_capacity(s.len() + 2);
o.push('"');
for c in s.chars() {
match c {
'"' => o.push_str("\\\""),
'\\' => o.push_str("\\\\"),
'\n' => o.push_str("\\n"),
_ => o.push(c),
}
}
o.push('"');
o
}
fn join_hex(v: &[u32]) -> String {
v.iter().map(|&x| hex(x)).collect::<Vec<_>>().join(", ")
}
fn join_jhex(v: &[u32]) -> String {
v.iter().map(|&x| jhex(x)).collect::<Vec<_>>().join(", ")
}
/// The three dialects of the memory-hard core.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum CoreDialect {
Metal,
Cuda,
OpenCl,
}
/// How the hash kernel gets dataset words. `Stored` is the honest kernel; the inline variants are the
/// shortcut measurements of MEMHARD.md section 2.2.
pub enum LoadSource<'a> {
Stored,
InlineClosed(u32, u32),
InlineMemhard(&'a MixParams),
}
fn log2_segments() -> usize {
CACHE_SEGMENTS.trailing_zeros() as usize
}
/// The memory-hard core as source text (`emitMemhardCore`). Every parameter is a literal.
pub fn emit_memhard_core(mp: &MixParams, dialect: CoreDialect) -> String {
let (u, fn_, cptr, wptr, lptr, lcptr) = match dialect {
CoreDialect::Metal => {
("uint", "inline", "device const uint*", "device uint*", "thread uint*", "const thread uint*")
}
CoreDialect::Cuda => ("uint32_t", "IGNEUM_HD", "const uint32_t*", "uint32_t*", "uint32_t*", "const uint32_t*"),
CoreDialect::OpenCl => {
("uint", "static inline", "__global const uint*", "__global uint*", "uint*", "const uint*")
}
};
let k = &mp.key;
let r = &mp.rot;
let m = &mp.mul;
let c = &mp.rc;
let mut s = String::with_capacity(6000);
s.push_str(&format!(
"// Memory-hard dataset core (MEMHARD.md). Cache: 2^{} words in 2^{} segments of {} chained ChaCha{} lines.\n",
CACHE_LOG2_WORDS,
log2_segments(),
CACHE_LINES_PER_SEGMENT,
CHACHA_ROUNDS
));
s.push_str("// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.\n");
s.push_str(&format!("#define MH_CACHE_LINE_MASK {}\n", hex(CACHE_LINE_MASK)));
s.push_str(&format!("#define MH_SEGMENT_LINES {}u\n", CACHE_LINES_PER_SEGMENT));
s.push_str("#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }\n");
s.push_str(&format!(
"{fn_} {u} mh_rotl({u} x, {u} n) {{ return (x << n) | (x >> (32u - n)); }} // n in 1..31 at every call site\n"
));
s.push('\n');
s.push_str(&format!("// y = ChaCha{CHACHA_ROUNDS} core(x) + x\n"));
s.push_str(&format!("{fn_} void mh_chacha_block({lcptr} x, {lptr} y) {{\n"));
s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) y[i] = x[i];\n"));
s.push_str(&format!(" for ({u} r = 0u; r < {}u; ++r) {{\n", CHACHA_ROUNDS / 2));
s.push_str(
" MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)\n",
);
s.push_str(
" MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)\n",
);
s.push_str(
" MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)\n",
);
s.push_str(
" MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)\n",
);
s.push_str(" }\n");
s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) y[i] += x[i];\n"));
s.push_str("}\n");
s.push('\n');
s.push_str(&format!(
"// One cache segment: {} chained lines written at cache[seg * {}]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.\n",
CACHE_LINES_PER_SEGMENT,
CACHE_LINES_PER_SEGMENT * 16
));
s.push_str(&format!("{fn_} void mh_cache_segment({wptr} cache, {u} seg) {{\n"));
s.push_str(&format!(" {u} prev[16]; {u} x[16]; {u} y[16];\n"));
s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) prev[i] = 0u;\n"));
s.push_str(&format!(" for ({u} j = 0u; j < MH_SEGMENT_LINES; ++j) {{\n"));
s.push_str(&format!(
" x[0] = {} ^ prev[0]; x[1] = {} ^ prev[1]; x[2] = {} ^ prev[2]; x[3] = {} ^ prev[3];\n",
hex(CHACHA_SIGMA[0]),
hex(CHACHA_SIGMA[1]),
hex(CHACHA_SIGMA[2]),
hex(CHACHA_SIGMA[3])
));
for i in 0..8 {
s.push_str(&format!(" x[{}] = {} ^ prev[{}];\n", 4 + i, hex(k[i]), 4 + i));
}
s.push_str(&format!(
" x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = {} ^ prev[14]; x[15] = {} ^ prev[15];\n",
hex(CACHE_TAG[0]),
hex(CACHE_TAG[1])
));
s.push_str(" mh_chacha_block(x, y);\n");
s.push_str(&format!(" {wptr} line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);\n"));
s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) {{ line[i] = y[i]; prev[i] = y[i]; }}\n"));
s.push_str(" }\n");
s.push_str("}\n");
s.push('\n');
s.push_str("// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.\n");
s.push_str(&format!("{fn_} void mh_mixer({lptr} s, {u} rk) {{\n"));
for i in 0..16 {
s.push_str(&format!(" s[{i}] = (s[{i}] ^ ({} + rk)) * {};\n", hex(c[i]), hex(m[i])));
}
let col = (0..4).map(|i| format!("{}u", r[i])).collect::<Vec<_>>().join(", ");
let dia = (4..8).map(|i| format!("{}u", r[i])).collect::<Vec<_>>().join(", ");
s.push_str(&format!(" MH_QR(s[0], s[4], s[8], s[12], {col}) MH_QR(s[1], s[5], s[9], s[13], {col})\n"));
s.push_str(&format!(" MH_QR(s[2], s[6], s[10], s[14], {col}) MH_QR(s[3], s[7], s[11], s[15], {col})\n"));
s.push_str(&format!(" MH_QR(s[0], s[5], s[10], s[15], {dia}) MH_QR(s[1], s[6], s[11], s[12], {dia})\n"));
s.push_str(&format!(" MH_QR(s[2], s[7], s[8], s[13], {dia}) MH_QR(s[3], s[4], s[9], s[14], {dia})\n"));
s.push_str("}\n");
s.push('\n');
s.push_str(&format!(
"// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of mixer + cache line s[0] & mask; final mixer.\n"
));
s.push_str(&format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n"));
for i in 0..8 {
s.push_str(&format!(" s[{i}] = {};\n", hex(k[i])));
}
for i in 0..8 {
s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(m[i]), hex(c[i])));
}
s.push_str(&format!(" for ({u} r = 0u; r < {ITEM_ROUNDS}u; ++r) {{\n"));
s.push_str(" mh_mixer(s, 0x9E3779B9u * (r + 1u));\n");
s.push_str(&format!(" {cptr} line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);\n"));
s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= line[i];\n"));
s.push_str(" }\n");
s.push_str(&format!(" mh_mixer(s, 0x9E3779B9u * {}u);\n", ITEM_ROUNDS + 1));
s.push_str("}\n");
s.push_str("// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.\n");
s.push_str(&format!(
"{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }}\n"
));
s
}
/// Metal library with the cache fill and dataset build kernels for one day key (`memhardMSL`, memhard.metal).
pub fn metal_memhard(mp: &MixParams) -> String {
let mut s = String::new();
s.push_str("#include <metal_stdlib>\n");
s.push_str("using namespace metal;\n");
s.push_str(&emit_memhard_core(mp, CoreDialect::Metal));
s.push('\n');
s.push_str(&format!("// One thread per segment (2^{} threads).\n", log2_segments()));
s.push_str(
"kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {\n",
);
s.push_str(" mh_cache_segment(cache, gid);\n");
s.push_str("}\n");
s.push_str("// One thread per 64-byte item (dataset words / 16 threads).\n");
s.push_str(
"kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n",
);
s.push_str(" uint gid [[thread_position_in_grid]]) {\n");
s.push_str(" uint s[16];\n");
s.push_str(" mh_item(cache, gid, s);\n");
s.push_str(" device uint* d = dataset + gid * 16u;\n");
s.push_str(" for (uint i = 0u; i < 16u; ++i) d[i] = s[i];\n");
s.push_str("}\n");
s
}
const DS_ELEM_BODY: &str = " x *= 0x9E3779B1u; x ^= x >> 15;\n x += d1;\n x *= 0x85EBCA77u; x ^= x >> 13;\n x *= 0xC2B2AE3Du; x ^= x >> 16;\n return x;\n}\n";
/// The Metal hash kernel (`generateMSL`, program.metal).
pub fn metal_program(p: &Program, dataset_log2: u32, source: LoadSource) -> String {
metal_program_impl(p, dataset_log2, source, false)
}
/// The header-bound Metal kernel (`program_bound.metal`, serve mode of proto-metal): `igneum_hash_bound` reads its
/// init words `I` from `constant uint* initw [[buffer(3)]]` (`bind::block_init_words`) instead of `SEEDW`. Same
/// instruction text as `igneum_hash`. Stored dataset only.
pub fn metal_program_bound(p: &Program, dataset_log2: u32) -> String {
metal_program_impl(p, dataset_log2, LoadSource::Stored, true)
}
fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: bool) -> String {
let mask = mask_for(dataset_log2);
let mut s = String::with_capacity(5000);
s.push_str("#include <metal_stdlib>\n");
s.push_str("using namespace metal;\n");
s.push('\n');
s.push_str(&format!("#define MASK {}\n", hex(mask)));
s.push_str(&format!("constant uint SEEDW[8] = {{ {} }};\n", join_hex(&p.seed)));
s.push('\n');
s.push_str("inline uint splitmix32(uint x) {\n");
s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n");
s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n");
s.push_str(" x ^= x >> 16;\n");
s.push_str(" return x;\n");
s.push_str("}\n");
s.push_str("inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31\n");
s.push_str("inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n");
s.push_str("inline uint ds_elem(uint i, uint d0, uint d1) {\n");
s.push_str(" uint x = i ^ d0;\n");
s.push_str(DS_ELEM_BODY);
s.push('\n');
if p.has_wide() {
s.push_str("#define WMASK (MASK & ~31u)\n\n");
}
let mut buffer0 = "device const uint* dataset [[buffer(0)]]";
if let LoadSource::InlineMemhard(mp) = &source {
s.push_str(&emit_memhard_core(mp, CoreDialect::Metal));
s.push('\n');
buffer0 = "device const uint* cache [[buffer(0)]]";
}
if bound {
s.push_str("// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.\n");
s.push_str(&format!("kernel void igneum_hash_bound({buffer0},\n"));
} else {
s.push_str(&format!("kernel void igneum_hash({buffer0},\n"));
}
s.push_str(" device ulong* out [[buffer(1)]],\n");
s.push_str(" constant uint& baseNonce [[buffer(2)]],\n");
if bound {
s.push_str(" constant uint* initw [[buffer(3)]],\n");
}
s.push_str(" uint gid [[thread_position_in_grid]]) {\n");
s.push_str(" uint nonce = baseNonce + gid;\n");
s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n");
if p.has_wide() {
s.push_str(" uint lane = gid & 31u;\n");
}
let iw = if bound { "initw" } else { "SEEDW" };
for i in 0..8 {
s.push_str(&format!(
" {{ uint x = nonce ^ {iw}[{i}]; x += 0x9e3779b9u * {}u; x = splitmix32(x); r{i} = x ^ {iw}[{}]; }}\n",
i + 1,
(i + 1) & 7
));
}
s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n"));
let word_index = |a: &str, wide: bool| -> String {
if wide {
format!("(simd_broadcast({a}, 0) & WMASK) + lane")
} else {
format!("{a} & MASK")
}
};
let fetch = |idx: String| -> String {
match &source {
LoadSource::Stored => format!("dataset[{idx}]"),
LoadSource::InlineClosed(d0, d1) => format!("ds_elem({idx}, {}, {})", hex(*d0), hex(*d1)),
LoadSource::InlineMemhard(_) => format!("mh_word(cache, {idx})"),
}
};
for (k, ins) in p.instrs.iter().enumerate() {
let d = format!("r{}", ins.dst);
let a = format!("r{}", ins.src);
let b = format!("r{}", ins.src2);
let line = match ins.op {
Op::Add => format!(
"{d} = {d} + {a} + select({}, {}, ((sel >> {}u) & 1u) != 0u);",
hex(ins.imm),
hex(ins.imm2),
ins.bit
),
Op::Sub => format!("{d} = {d} - {a};"),
Op::Mul => format!("{d} = {d} * {a};"),
Op::MulHi => format!("{d} = mulhi({d}, {a});"),
Op::Xor => format!("{d} = {d} ^ {a};"),
Op::Or => format!("{d} = {d} | {a};"),
Op::Rotl => format!("{d} = rotl_imm({d}, {}u);", ins.rot),
Op::Rotr => format!("{d} = rotr_var({d}, {a});"),
Op::Mad => format!("{d} = {a} * {b} + {d};"),
Op::Shfl => format!("{d} = {d} ^ simd_shuffle_xor({a}, (ushort){});", ins.mask),
Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&a, false))),
Op::WLoad => format!("{d} = {d} ^ {};", fetch(word_index(&a, true))),
};
s.push_str(&format!(" {line} // {k}\n"));
}
s.push_str(" }\n");
s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n");
s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n");
s.push_str("}\n");
s
}
/// The Metal closed-form fill kernel (`fillMSL`).
pub const METAL_FILL: &str = "#include <metal_stdlib>\nusing namespace metal;\ninline uint ds_elem(uint i, uint d0, uint d1) {\n uint x = i ^ d0;\n x *= 0x9E3779B1u; x ^= x >> 15;\n x += d1;\n x *= 0x85EBCA77u; x ^= x >> 13;\n x *= 0xC2B2AE3Du; x ^= x >> 16;\n return x;\n}\nkernel void igneum_fill(device uint* dataset [[buffer(0)]],\n constant uint2& day [[buffer(1)]],\n uint gid [[thread_position_in_grid]]) {\n dataset[gid] = ds_elem(gid, day.x, day.y);\n}";
fn generated_by(seed: &str) -> String {
format!("// Generated by igneum-pow export (generator v{GENERATOR_VERSION}) for seed \"{seed}\". Do not edit by hand.\n")
}
fn hex_bytes(b: &[u8]) -> String {
b.iter().map(|x| format!("{x:02x}")).collect()
}
fn init_line(p: &Program, u: &str, i: usize) -> String {
let addc = 0x9e3779b9u32.wrapping_mul(i as u32 + 1);
format!(
" {{ {u} x = nonce ^ {}; x += {}; x = splitmix32(x); r{i} = x ^ {}; }} // SEEDW[{i}], 0x9e3779b9u * {}u, SEEDW[{}]\n",
hex(p.seed[i]),
hex(addc),
hex(p.seed[(i + 1) & 7]),
i + 1,
(i + 1) & 7
)
}
/// The instruction lines of the CUDA hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`).
fn cuda_instr_lines(p: &Program) -> String {
let mut s = String::with_capacity(6000);
for (k, ins) in p.instrs.iter().enumerate() {
let d = format!("r{}", ins.dst);
let a = format!("r{}", ins.src);
let b = format!("r{}", ins.src2);
let line = match ins.op {
// Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well.
Op::Add => format!(
"{d} = {d} + {a} + ((((sel >> {}u) & 1u) != 0u) ? {} : {});",
ins.bit,
hex(ins.imm2),
hex(ins.imm)
),
Op::Sub => format!("{d} = {d} - {a};"),
Op::Mul => format!("{d} = {d} * {a};"),
Op::MulHi => format!("{d} = __umulhi({d}, {a});"),
Op::Xor => format!("{d} = {d} ^ {a};"),
Op::Or => format!("{d} = {d} | {a};"),
Op::Rotl => format!("{d} = rotl_imm({d}, {}u);", ins.rot),
Op::Rotr => format!("{d} = rotr_var({d}, {a});"),
Op::Mad => format!("{d} = {a} * {b} + {d};"),
Op::Shfl => format!("{d} = {d} ^ __shfl_xor_sync(0xffffffffu, {a}, {});", ins.mask),
Op::Load => format!("{d} = {d} ^ ds[{a} & mask];"),
Op::WLoad => format!("{d} = {d} ^ ds[(__shfl_sync(0xffffffffu, {a}, 0) & wmask) + lane];"),
};
s.push_str(&format!(" {line} // {k} {}\n", ins.op.name()));
}
s
}
/// The CUDA kernel (`generateCUDA`, kernel.cu). `memhard` is `None` for a closed-form pack.
pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String {
let mut s = String::with_capacity(9000);
s.push_str(&generated_by(&p.seed_string));
s.push_str(
"// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).\n",
);
s.push_str("// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.\n");
s.push_str("#include <cuda_runtime.h>\n");
s.push_str("#include <cstdint>\n");
s.push_str("#include \"program.h\"\n");
if memhard.is_some() {
s.push_str("#include \"memhard.h\"\n");
}
s.push('\n');
s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n");
s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n");
s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n");
s.push_str(" x ^= x >> 16;\n");
s.push_str(" return x;\n");
s.push_str("}\n");
s.push_str("// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.\n");
s.push_str("__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }\n");
s.push_str("// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.\n");
s.push_str("__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n");
s.push_str("__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {\n");
s.push_str(" uint32_t x = i ^ d0;\n");
s.push_str(DS_ELEM_BODY);
s.push('\n');
if memhard.is_none() {
s.push_str("// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.\n");
s.push_str("__global__ void igneum_fill(uint32_t* ds, uint32_t n, uint32_t d0, uint32_t d1) {\n");
s.push_str(" uint32_t i = blockIdx.x * blockDim.x + threadIdx.x;\n");
s.push_str(" if (i < n) ds[i] = ds_elem(i, d0, d1);\n");
s.push_str("}\n");
s.push('\n');
} else {
s.push_str(
"// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.\n",
);
s.push_str(
"// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.\n",
);
s.push_str("__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {\n");
s.push_str(" uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;\n");
s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n");
s.push_str("}\n");
s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n");
s.push_str(" uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;\n");
s.push_str(" if (t < nItems) {\n");
s.push_str(" uint32_t s[16];\n");
s.push_str(" mh_item(cache, t, s);\n");
s.push_str(" uint32_t* d = ds + (size_t)t * 16u;\n");
s.push_str(" for (uint32_t i = 0u; i < 16u; ++i) d[i] = s[i];\n");
s.push_str(" }\n");
s.push_str("}\n");
s.push('\n');
}
s.push_str("// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every\n");
s.push_str("// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a\n");
s.push_str("// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.\n");
s.push_str("__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask) {\n");
s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n");
s.push_str(" uint32_t nonce = baseNonce + gid;\n");
s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n");
if p.has_wide() {
s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n");
}
for i in 0..8 {
s.push_str(&init_line(p, "uint32_t", i));
}
s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n"));
s.push_str(&cuda_instr_lines(p));
s.push_str(" }\n");
s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n");
s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n");
s.push_str("}\n");
s.push('\n');
s.push_str("// Host-side launch wrappers. Declared in program.h, called from host.cu.\n");
if memhard.is_none() {
s.push_str("cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1) {\n");
s.push_str(" if (nWords == 0u) return cudaErrorInvalidValue;\n");
s.push_str(" uint32_t block = 256u;\n");
s.push_str(" uint32_t grid = (nWords + block - 1u) / block;\n");
s.push_str(" igneum_fill<<<grid, block>>>(ds, nWords, d0, d1);\n");
s.push_str(" return cudaGetLastError();\n");
s.push_str("}\n");
s.push('\n');
} else {
s.push_str("cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {\n");
s.push_str(" if (nSegments == 0u) return cudaErrorInvalidValue;\n");
s.push_str(" uint32_t block = 256u;\n");
s.push_str(" uint32_t grid = (nSegments + block - 1u) / block;\n");
s.push_str(" igneum_cache_fill<<<grid, block>>>(cache, nSegments);\n");
s.push_str(" return cudaGetLastError();\n");
s.push_str("}\n");
s.push('\n');
s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n");
s.push_str(" if (nItems == 0u) return cudaErrorInvalidValue;\n");
s.push_str(" uint32_t block = 256u;\n");
s.push_str(" uint32_t grid = (nItems + block - 1u) / block;\n");
s.push_str(" igneum_build<<<grid, block>>>(ds, cache, nItems);\n");
s.push_str(" return cudaGetLastError();\n");
s.push_str("}\n");
s.push('\n');
}
s.push_str(
"cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n",
);
s.push_str(" uint32_t nonces, uint32_t blockWarps) {\n");
s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n");
s.push_str(" uint32_t block = 32u * blockWarps;\n");
s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n");
s.push_str(" igneum_hash<<<nonces / block, block>>>(ds, out, baseNonce, mask);\n");
s.push_str(" return cudaGetLastError();\n");
s.push_str("}\n");
s.push('\n');
s.push_str("cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {\n");
s.push_str(" cudaFuncAttributes attr;\n");
s.push_str(" cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);\n");
s.push_str(" if (e != cudaSuccess) return e;\n");
s.push_str(" *numRegs = attr.numRegs;\n");
s.push_str(" return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);\n");
s.push_str("}\n");
s
}
/// `kernel_bound.cu`: the header-bound CUDA hash kernel for the serve mode of proto-cuda. A standalone
/// translation unit (compiled next to kernel.cu, which keeps the cache-fill and build wrappers): the init words
/// `I` arrive by value in `IgneumInitWords` (`bind::block_init_words`), the instruction text is that of
/// `igneum_hash`. Declarations for the host are at the top of the file.
pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String {
let mut s = String::with_capacity(9000);
s.push_str(&generated_by(&p.seed_string));
s.push_str(
"// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.\n",
);
s.push_str("// Host declarations (also in program_bound.h if present):\n");
s.push_str("// struct IgneumInitWords { uint32_t w[8]; };\n");
s.push_str("// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n");
s.push_str(
"// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);\n",
);
s.push_str("// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);\n");
s.push_str("#include <cuda_runtime.h>\n");
s.push_str("#include <cstdint>\n");
s.push_str("#include \"program.h\"\n");
s.push('\n');
s.push_str("struct IgneumInitWords { uint32_t w[8]; };\n");
s.push('\n');
s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n");
s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n");
s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n");
s.push_str(" x ^= x >> 16;\n");
s.push_str(" return x;\n");
s.push_str("}\n");
s.push_str("__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }\n");
s.push_str("__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n");
s.push('\n');
let _ = memhard; // the bound kernel reads the stored dataset in both constructions
s.push_str("__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw) {\n");
s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n");
s.push_str(" uint32_t nonce = baseNonce + gid;\n");
s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n");
if p.has_wide() {
s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n");
}
for i in 0..8 {
s.push_str(&format!(
" {{ uint32_t x = nonce ^ iw.w[{i}]; x += 0x9e3779b9u * {}u; x = splitmix32(x); r{i} = x ^ iw.w[{}]; }}\n",
i + 1,
(i + 1) & 7
));
}
s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n"));
s.push_str(&cuda_instr_lines(p));
s.push_str(" }\n");
s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n");
s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n");
s.push_str("}\n");
s.push('\n');
s.push_str(
"cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n",
);
s.push_str(" IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps) {\n");
s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n");
s.push_str(" uint32_t block = 32u * blockWarps;\n");
s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n");
s.push_str(" igneum_hash_bound<<<nonces / block, block>>>(ds, out, baseNonce, mask, iw);\n");
s.push_str(" return cudaGetLastError();\n");
s.push_str("}\n");
s.push('\n');
s.push_str("cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {\n");
s.push_str(" cudaFuncAttributes attr;\n");
s.push_str(" cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);\n");
s.push_str(" if (e != cudaSuccess) return e;\n");
s.push_str(" *numRegs = attr.numRegs;\n");
s.push_str(" return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);\n");
s.push_str("}\n");
s
}
/// The instruction lines of the OpenCL hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`).
fn opencl_instr_lines(p: &Program) -> String {
let mut s = String::with_capacity(6000);
for (k, ins) in p.instrs.iter().enumerate() {
let d = format!("r{}", ins.dst);
let a = format!("r{}", ins.src);
let b = format!("r{}", ins.src2);
let line = match ins.op {
Op::Add => format!(
"{d} = {d} + {a} + ((((sel >> {}u) & 1u) != 0u) ? {} : {});",
ins.bit,
hex(ins.imm2),
hex(ins.imm)
),
Op::Sub => format!("{d} = {d} - {a};"),
Op::Mul => format!("{d} = {d} * {a};"),
Op::MulHi => format!("{d} = mul_hi({d}, {a});"),
Op::Xor => format!("{d} = {d} ^ {a};"),
Op::Or => format!("{d} = {d} | {a};"),
Op::Rotl => format!("{d} = rotl_imm({d}, {}u);", ins.rot),
Op::Rotr => format!("{d} = rotr_var({d}, {a});"),
Op::Mad => format!("{d} = {a} * {b} + {d};"),
Op::Shfl => format!("{{ uint t_; IGNEUM_SHFL_XOR(t_, {a}, {}u); {d} = {d} ^ t_; }}", ins.mask),
Op::Load => format!("{d} = {d} ^ ds[{a} & mask];"),
Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[(t_ & wmask) + lane]; }}"),
};
s.push_str(&format!(" {line} // {k} {}\n", ins.op.name()));
}
s
}
/// `kernel_bound.cl`: `kernel.cl` plus the header-bound kernel `igneum_hash_bound`, whose init words come from a
/// fifth argument (`__global const uint* initw`, 8 words, `bind::block_init_words`). One source file so the serve
/// mode of proto-opencl/host.c builds cache fill, dataset build and the bound hash from it at runtime.
pub fn opencl_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String {
let mut s = opencl_kernel(p, memhard);
s.push('\n');
s.push_str(
"// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.\n",
);
s.push_str("IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw) {\n");
s.push_str(" uint gid = (uint)get_global_id(0);\n");
s.push_str(" uint lid = (uint)get_local_id(0);\n");
s.push_str(" uint nonce = baseNonce + gid;\n");
s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n");
s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n");
s.push_str("#if IGNEUM_EXCHANGE == 0\n");
s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n");
s.push_str(" uint xk = 0u;\n");
s.push_str("#else\n");
s.push_str(" (void)lid;\n");
s.push_str("#endif\n");
if p.has_wide() {
s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n");
}
for i in 0..8 {
s.push_str(&format!(
" {{ uint x = nonce ^ iw{i}; x += 0x9e3779b9u * {}u; x = splitmix32(x); r{i} = x ^ iw{}; }}\n",
i + 1,
(i + 1) & 7
));
}
s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n"));
s.push_str(&opencl_instr_lines(p));
s.push_str(" }\n");
s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n");
s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n");
s.push_str("}\n");
s
}
/// The OpenCL C 1.2 kernel (`generateOpenCL`, kernel.cl).
pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String {
let mut s = String::with_capacity(14000);
s.push_str(&generated_by(&p.seed_string));
s.push_str("// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).\n");
s.push_str("// Built from source at runtime by proto-opencl/host.c, which passes these defines:\n");
s.push_str("// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)\n");
s.push_str(
"// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)\n",
);
s.push_str("// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32\n");
s.push_str("// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition\n");
s.push_str("// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;\n");
s.push_str("// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.\n");
s.push_str("#ifndef IGNEUM_GROUP\n#define IGNEUM_GROUP 32\n#endif\n");
s.push_str("#ifndef IGNEUM_EXCHANGE\n#define IGNEUM_EXCHANGE 0\n#endif\n");
s.push_str("#ifdef __OPENCL_VERSION__\n");
s.push_str("#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))\n");
s.push_str("#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]\n");
s.push_str("#if IGNEUM_EXCHANGE == 1\n");
s.push_str("#ifdef cl_khr_subgroups\n#pragma OPENCL EXTENSION cl_khr_subgroups : enable\n#endif\n");
s.push_str("#ifdef cl_khr_subgroup_shuffle\n#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable\n#endif\n");
s.push_str("#elif IGNEUM_EXCHANGE == 2\n#pragma OPENCL EXTENSION cl_intel_subgroups : enable\n#endif\n");
s.push_str("#else\n");
s.push_str("// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.\n");
s.push_str("#include \"emu_opencl.h\"\n#endif\n");
s.push('\n');
s.push_str("#if IGNEUM_EXCHANGE == 1\n");
s.push_str("#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))\n");
s.push_str("#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)\n");
s.push_str("#elif IGNEUM_EXCHANGE == 2\n");
s.push_str("#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))\n");
s.push_str("#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)\n");
s.push_str("#else\n");
s.push_str("// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per\n");
s.push_str("// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane\n");
s.push_str("// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's\n");
s.push_str(
"// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.\n",
);
s.push_str("#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }\n");
s.push_str("#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }\n");
s.push_str("#endif\n");
s.push('\n');
s.push_str("static inline uint splitmix32(uint x) {\n");
s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n");
s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n");
s.push_str(" x ^= x >> 16;\n");
s.push_str(" return x;\n");
s.push_str("}\n");
s.push_str("// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.\n");
s.push_str("static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }\n");
s.push_str("// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.\n");
s.push_str("static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }\n");
s.push_str("static inline uint ds_elem(uint i, uint d0, uint d1) {\n");
s.push_str(" uint x = i ^ d0;\n");
s.push_str(DS_ELEM_BODY);
s.push('\n');
if let Some(mp) = memhard {
s.push_str(&emit_memhard_core(mp, CoreDialect::OpenCl));
s.push('\n');
s.push_str("// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.\n");
s.push_str("// The same constants as memhard.h in this pack (one emitter, three dialects).\n");
s.push_str("__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {\n");
s.push_str(" uint seg = (uint)get_global_id(0);\n");
s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n");
s.push_str("}\n");
s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {\n");
s.push_str(" uint t = (uint)get_global_id(0);\n");
s.push_str(" if (t < nItems) {\n");
s.push_str(" uint s[16];\n");
s.push_str(" mh_item(cache, t, s);\n");
s.push_str(" __global uint* d = ds + ((ulong)t * 16u);\n");
s.push_str(" for (uint i = 0u; i < 16u; ++i) d[i] = s[i];\n");
s.push_str(" }\n");
s.push_str("}\n");
s.push('\n');
} else {
s.push_str("// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.\n");
s.push_str("__kernel void igneum_fill(__global uint* ds, uint n, uint d0, uint d1) {\n");
s.push_str(" uint i = (uint)get_global_id(0);\n");
s.push_str(" if (i < n) ds[i] = ds_elem(i, d0, d1);\n");
s.push_str("}\n");
s.push('\n');
}
s.push_str("// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the\n");
s.push_str("// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and\n");
s.push_str("// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).\n");
s.push_str("IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask) {\n");
s.push_str(" uint gid = (uint)get_global_id(0);\n");
s.push_str(" uint lid = (uint)get_local_id(0);\n");
s.push_str(" uint nonce = baseNonce + gid;\n");
s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n");
s.push_str("#if IGNEUM_EXCHANGE == 0\n");
s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n");
s.push_str(" uint xk = 0u;\n");
s.push_str("#else\n");
s.push_str(" (void)lid;\n");
s.push_str("#endif\n");
if p.has_wide() {
s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n");
}
for i in 0..8 {
s.push_str(&init_line(p, "uint", i));
}
s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n"));
s.push_str(&opencl_instr_lines(p));
s.push_str(" }\n");
s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n");
s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n");
s.push_str("}\n");
s.push('\n');
s.push_str("#if IGNEUM_EXCHANGE != 0\n");
s.push_str("// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the\n");
s.push_str("// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a\n");
s.push_str("// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.\n");
s.push_str("IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {\n");
s.push_str(" if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }\n");
s.push_str("}\n");
s.push_str("#endif\n");
s
}
fn mask_for(dataset_log2: u32) -> u32 {
if dataset_log2 >= 32 {
u32::MAX
} else {
(1u32 << dataset_log2) - 1
}
}
const STDINT_BLOCK: &str = "#ifdef __cplusplus\n#include <cstdint>\n#else\n#include <stdint.h>\n#endif\n";
/// program.h (`generateProgramHeader`).
pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String {
let key = &ds.key;
let dataset_log2 = ds.log2_words;
let memhard = ds.memhard().map(|m| &m.params);
let mask = mask_for(dataset_log2);
let mut s = String::with_capacity(2600);
s.push_str(&generated_by(&p.seed_string));
s.push_str("// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.\n");
s.push_str("// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.\n");
s.push_str("#pragma once\n");
s.push_str(STDINT_BLOCK);
s.push_str("#ifndef IGNEUM_NO_CUDA\n#include <cuda_runtime.h>\n#endif\n");
s.push('\n');
s.push_str(&format!("#define IGNEUM_SEED_STRING {}\n", jstr(&p.seed_string)));
s.push_str(&format!("#define IGNEUM_SEED_BYTES_HEX {}\n", jstr(&hex_bytes(&p.seed_bytes))));
s.push_str(&format!("#define IGNEUM_GENERATOR {}\n", p.generator));
s.push_str(&format!("#define IGNEUM_PROGRAM_ATTEMPT {}\n", p.attempt));
s.push_str(&format!("#define IGNEUM_PROGRAM_ID {}\n", hex64(p.program_id())));
s.push_str(&format!("#define IGNEUM_DAY_STRING {}\n", jstr(day)));
s.push_str(&format!("#define IGNEUM_DAY_BYTES_HEX {}\n", jstr(&hex_bytes(&ds.key_bytes))));
s.push_str(&format!("#define IGNEUM_DAY0 {}\n", hex(key[0])));
s.push_str(&format!("#define IGNEUM_DAY1 {}\n", hex(key[1])));
s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n"));
s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask)));
s.push_str("#define IGNEUM_LANES 32\n");
s.push_str(&format!("#define IGNEUM_ITERATIONS {ITERATIONS}\n"));
s.push_str(&format!("#define IGNEUM_INSTR_COUNT {INSTR_COUNT}\n"));
s.push_str(&format!("#define IGNEUM_LOADS_PER_HASH {}\n", p.loads_per_hash()));
s.push_str(&format!("#define IGNEUM_WIDE_LOADS_PER_HASH {}\n", p.wide_loads_per_hash()));
s.push_str(&format!("#define IGNEUM_OP_MIX {}\n", jstr(&p.op_mix())));
s.push_str("// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)\n");
s.push_str(&format!("#define IGNEUM_DATASET_MODE {}\n", if memhard.is_some() { 1 } else { 0 }));
s.push('\n');
s.push_str(&format!("#define IGNEUM_SEEDW_INIT {{ {} }}\n", join_hex(&p.seed)));
if let Some(mp) = memhard {
s.push_str(&format!("#define IGNEUM_KEY_INIT {{ {} }}\n", join_hex(&mp.key)));
s.push_str(&format!("#define IGNEUM_CACHE_LOG2_WORDS {CACHE_LOG2_WORDS}\n"));
s.push_str(&format!("#define IGNEUM_CACHE_SEGMENT_LOG2_LINES {CACHE_SEGMENT_LOG2_LINES}\n"));
s.push_str(&format!("#define IGNEUM_CACHE_SEGMENTS {CACHE_SEGMENTS}u\n"));
s.push_str(&format!("#define IGNEUM_ITEM_ROUNDS {ITEM_ROUNDS}\n"));
s.push_str(&format!(
"#define IGNEUM_MIX_ROT_INIT {{ {} }}\n",
mp.rot.iter().map(|r| format!("{r}u")).collect::<Vec<_>>().join(", ")
));
s.push_str(&format!("#define IGNEUM_MIX_MUL_INIT {{ {} }}\n", join_hex(&mp.mul)));
s.push_str(&format!("#define IGNEUM_MIX_RC_INIT {{ {} }}\n", join_hex(&mp.rc)));
s.push('\n');
s.push_str("#ifndef IGNEUM_NO_CUDA\n");
s.push_str("// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().\n");
s.push_str("cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);\n");
s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);\n");
} else {
s.push_str("#ifndef IGNEUM_NO_CUDA\n");
s.push_str("// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().\n");
s.push_str("cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);\n");
}
s.push_str(
"cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n",
);
s.push_str(" uint32_t nonces, uint32_t blockWarps);\n");
s.push_str("cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);\n");
s.push_str("#endif\n");
s
}
/// memhard.h (`generateMemhardHeader`): the core in CUDA C++, compiled for host and device.
pub fn cuda_memhard_header(p: &Program, mp: &MixParams) -> String {
let mut s = String::with_capacity(6000);
s.push_str(&generated_by(&p.seed_string));
s.push_str("// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.\n");
s.push_str(
"// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).\n",
);
s.push_str("// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.\n");
s.push_str("#pragma once\n");
s.push_str(STDINT_BLOCK);
s.push_str("#if defined(__CUDACC__)\n");
s.push_str("#define IGNEUM_HD __host__ __device__ __forceinline__\n");
s.push_str("#elif defined(_MSC_VER) && !defined(__cplusplus)\n");
s.push_str("#define IGNEUM_HD static __inline\n");
s.push_str("#else\n");
s.push_str("#define IGNEUM_HD static inline\n");
s.push_str("#endif\n");
s.push_str(&emit_memhard_core(mp, CoreDialect::Cuda));
s
}
/// The self-test values a pack carries beside the 96 hashes.
#[derive(Clone, Debug, Default, PartialEq, Eq)]
pub struct PackVectors {
/// dataset[0..15]
pub head: Vec<u32>,
/// dataset[MASK]
pub last: u32,
/// 64 sampled dataset indices and their values
pub sample_idx: Vec<u32>,
pub sample_val: Vec<u32>,
/// cache[0..15] (memory-hard only)
pub cache_head: Vec<u32>,
/// the last cache line (memory-hard only)
pub cache_last: Vec<u32>,
/// FNV-1a 64 over the whole cache (memory-hard only)
pub cache_fnv: u64,
}
/// The base nonces of the three vector warps every pack carries.
pub const PACK_VECTOR_BASES: [u32; 3] = [0, 4096, 1_000_000];
/// The 64 sampled dataset indices: SplitMix64 seeded with "mhsample", low 32 bits masked.
pub fn sample_indices(mask: u32) -> Vec<u32> {
let mut sr = SplitMix64::new(0x6d68_7361_6d70_6c65);
(0..64).map(|_| (sr.next() as u32) & mask).collect()
}
/// vectors.h (`generateVectorsHeader`).
pub fn vectors_header(
p: &Program,
bases: &[u32],
outs: &[[u64; 32]],
v: &PackVectors,
mask: u32,
source: &str,
memhard: bool,
) -> String {
let mut s = String::with_capacity(6000);
s.push_str(&generated_by(&p.seed_string));
s.push_str(&format!("// Expected outputs: {source}\n"));
s.push_str("#pragma once\n");
s.push_str(STDINT_BLOCK);
s.push('\n');
s.push_str(&format!("#define IGNEUM_VEC_WARPS {}\n", bases.len()));
s.push_str(&format!(
"static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = {{ {} }};\n",
bases.iter().map(|b| format!("{b}u")).collect::<Vec<_>>().join(", ")
));
s.push_str("static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {\n");
for (i, o) in outs.iter().enumerate() {
s.push_str(&format!(" {{ // base nonce {}\n", bases[i]));
for row in 0..4 {
s.push_str(" ");
s.push_str(&(0..8).map(|c| hex64(o[row * 8 + c])).collect::<Vec<_>>().join(", "));
s.push_str(if row == 3 { "\n" } else { ",\n" });
}
s.push_str(if i == outs.len() - 1 { " }\n" } else { " },\n" });
}
s.push_str("};\n");
s.push('\n');
s.push_str(&format!("// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] ({mask}).\n"));
s.push_str("static const uint32_t IGNEUM_DS_HEAD[16] = {\n");
s.push_str(&format!(" {},\n", join_hex(&v.head[..8])));
s.push_str(&format!(" {}\n", join_hex(&v.head[8..16])));
s.push_str("};\n");
s.push_str(&format!("static const uint32_t IGNEUM_DS_LAST_INDEX = {mask}u;\n"));
s.push_str(&format!("static const uint32_t IGNEUM_DS_LAST = {};\n", hex(v.last)));
s.push_str("// 64 sampled dataset words (index, value) computed on the Mac.\n");
s.push_str(&format!("#define IGNEUM_DS_SAMPLES {}\n", v.sample_idx.len()));
s.push_str("static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {\n");
s.push_str(&format!(" {}\n", v.sample_idx.iter().map(|i| format!("{i}u")).collect::<Vec<_>>().join(", ")));
s.push_str("};\n");
s.push_str("static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {\n");
s.push_str(&format!(" {}\n", join_hex(&v.sample_val)));
s.push_str("};\n");
if memhard {
s.push_str(&format!(
"// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^{CACHE_LOG2_WORDS} words.\n"
));
s.push_str("static const uint32_t IGNEUM_CACHE_HEAD[16] = {\n");
s.push_str(&format!(" {},\n", join_hex(&v.cache_head[..8])));
s.push_str(&format!(" {}\n", join_hex(&v.cache_head[8..16])));
s.push_str("};\n");
s.push_str("static const uint32_t IGNEUM_CACHE_LAST[16] = {\n");
s.push_str(&format!(" {},\n", join_hex(&v.cache_last[..8])));
s.push_str(&format!(" {}\n", join_hex(&v.cache_last[8..16])));
s.push_str("};\n");
s.push_str(&format!("static const uint64_t IGNEUM_CACHE_FNV64 = {};\n", hex64(v.cache_fnv)));
}
s
}
/// program.json (`generateProgramJSON`). Valid JSON (see the module note about the `"item"` line).
pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String {
let key = &ds.key;
let dataset_log2 = ds.log2_words;
let memhard = ds.memhard().map(|m| &m.params);
let mask = mask_for(dataset_log2);
let mut s = String::with_capacity(14000);
s.push_str("{\n");
s.push_str(" \"format\": \"igneum-program-pack-3\",\n");
s.push_str(&format!(" \"generator\": {},\n", p.generator));
s.push_str(&format!(" \"attempt\": {},\n", p.attempt));
s.push_str(&format!(" \"program_id\": {},\n", jhex64(p.program_id())));
s.push_str(" \"program_id_derivation\": \"FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32\",\n");
s.push_str(&format!(
" \"dataset_mode\": {},\n",
jstr(if memhard.is_some() { "memory-hard" } else { "closed-form" })
));
s.push_str(&format!(" \"seed\": {},\n", jstr(&p.seed_string)));
s.push_str(&format!(" \"seed_bytes\": {},\n", jstr(&hex_bytes(&p.seed_bytes))));
s.push_str(&format!(" \"seed_words\": [{}],\n", join_jhex(&p.seed)));
s.push_str(" \"seed_derivation\": \"seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32\",\n");
s.push_str(&format!(" \"generator_rule\": \"version {GENERATOR_VERSION}: exactly {LOAD_SLOTS} load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried\",\n"));
s.push_str(" \"lanes\": 32,\n");
s.push_str(" \"registers\": 8,\n");
s.push_str(&format!(" \"iterations\": {ITERATIONS},\n"));
s.push_str(&format!(" \"instruction_count\": {INSTR_COUNT},\n"));
s.push_str(&format!(" \"loads_per_hash\": {},\n", p.loads_per_hash()));
s.push_str(&format!(
" \"op_mix\": {{{}}},\n",
p.histogram().iter().map(|(n, c)| format!("{}: {c}", jstr(n))).collect::<Vec<_>>().join(", ")
));
s.push_str(" \"register_init\": \"for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]\",\n");
s.push_str(" \"splitmix32\": \"x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16\",\n");
s.push_str(
" \"iteration\": \"sel = r0 sampled once at the top of each iteration, then all instructions in order\",\n",
);
s.push_str(" \"output\": \"lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo\",\n");
s.push_str(" \"op_semantics\": {\n");
s.push_str(" \"add\": \"dst = dst + src + (bit `bit` of sel ? imm2 : imm)\",\n");
s.push_str(" \"sub\": \"dst = dst - src\",\n");
s.push_str(" \"mul\": \"dst = dst * src (low 32)\",\n");
s.push_str(" \"mulhi\": \"dst = high 32 bits of dst * src\",\n");
s.push_str(" \"xor\": \"dst = dst ^ src\",\n");
s.push_str(" \"or\": \"dst = dst | src\",\n");
s.push_str(" \"rotl\": \"dst = rotl(dst, rot), rot in 1..31\",\n");
s.push_str(" \"rotr\": \"dst = rotr(dst, src & 31)\",\n");
s.push_str(" \"mad\": \"dst = src * src2 + dst\",\n");
s.push_str(
" \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n",
);
s.push_str(" \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n");
s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\"\n");
s.push_str(" },\n");
s.push_str(" \"dataset\": {\n");
s.push_str(&format!(" \"log2_words\": {dataset_log2},\n"));
s.push_str(&format!(" \"bytes\": {},\n", 1u64 << (dataset_log2 as u64 + 2)));
s.push_str(&format!(" \"mask\": {},\n", jhex(mask)));
s.push_str(&format!(" \"day\": {},\n", jstr(day)));
s.push_str(&format!(" \"day_bytes\": {},\n", jstr(&hex_bytes(&ds.key_bytes))));
s.push_str(" \"day_words_from\": \"seed_words_from_bytes(day_bytes)\",\n");
s.push_str(&format!(" \"d0\": {},\n", jhex(key[0])));
s.push_str(&format!(" \"d1\": {},\n", jhex(key[1])));
if let Some(mp) = memhard {
s.push_str(" \"mode\": \"memory-hard\",\n");
s.push_str(" \"spec\": \"proto-metal/MEMHARD.md\",\n");
s.push_str(&format!(" \"key\": [{}],\n", join_jhex(&mp.key)));
s.push_str(" \"key_derivation\": \"the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]\",\n");
s.push_str(&format!(
" \"cache\": {{\"log2_words\": {CACHE_LOG2_WORDS}, \"bytes\": {}, \"line_words\": 16, \"segment_lines\": {CACHE_LINES_PER_SEGMENT}, \"segments\": {CACHE_SEGMENTS}, \"block\": \"ChaCha{CHACHA_ROUNDS} core + feed-forward, rotations 16 12 8 7\", \"sigma\": [{}], \"tag\": [{}], \"chain\": \"in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0\"}},\n",
CACHE_WORDS as u64 * 4,
join_jhex(&CHACHA_SIGMA),
join_jhex(&CACHE_TAG)
));
s.push_str(&format!(
" \"mixer\": {{\"draw\": \"SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())\", \"rot\": [{}], \"mul\": [{}], \"rc\": [{}], \"round\": \"for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]\", \"quarter_round\": \"a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)\"}},\n",
mp.rot.iter().map(|r| r.to_string()).collect::<Vec<_>>().join(", "),
join_jhex(&mp.mul),
join_jhex(&mp.rc)
));
// The Swift writes jhex(cacheLineMask) here, which breaks the JSON. We write the bare literal.
s.push_str(&format!(
" \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..{}: s = M_r(s); line = s[0] & 0x{:08x}; s[i] ^= cache[line * 16 + i]; then s = M_{ITEM_ROUNDS}(s); item(t) = s\",\n",
ITEM_ROUNDS - 1,
CACHE_LINE_MASK
));
s.push_str(" \"word\": \"dataset[w] = item(w >> 4)[w & 15]\"\n");
} else {
s.push_str(" \"mode\": \"closed-form\",\n");
s.push_str(" \"formula\": \"x = i ^ d0; x *= 0x9E3779B1; x ^= x>>15; x += d1; x *= 0x85EBCA77; x ^= x>>13; x *= 0xC2B2AE3D; x ^= x>>16 (all mod 2^32)\"\n");
}
s.push_str(" },\n");
s.push_str(" \"instructions\": [\n");
let n = p.instrs.len();
for (k, ins) in p.instrs.iter().enumerate() {
s.push_str(&format!(
" {{\"i\": {k}, \"op\": {}, \"dst\": {}, \"src\": {}, \"src2\": {}, \"imm\": {}, \"imm2\": {}, \"rot\": {}, \"bit\": {}, \"mask\": {}}}",
jstr(ins.op.name()),
ins.dst,
ins.src,
ins.src2,
jhex(ins.imm),
jhex(ins.imm2),
ins.rot,
ins.bit,
ins.mask
));
s.push_str(if k == n - 1 { "\n" } else { ",\n" });
}
s.push_str(" ]\n}\n");
s
}
/// vectors.json (`generateVectorsJSON`).
pub fn vectors_json(
p: &Program,
day: &str,
dataset_log2: u32,
bases: &[u32],
outs: &[[u64; 32]],
v: &PackVectors,
mask: u32,
source: &str,
memhard: bool,
) -> String {
let mut s = String::with_capacity(6500);
s.push_str("{\n");
s.push_str(&format!(" \"seed\": {},\n", jstr(&p.seed_string)));
s.push_str(&format!(" \"day\": {},\n", jstr(day)));
s.push_str(&format!(" \"dataset_mode\": {},\n", jstr(if memhard { "memory-hard" } else { "closed-form" })));
s.push_str(&format!(" \"dataset_log2_words\": {dataset_log2},\n"));
s.push_str(&format!(" \"mask\": {},\n", jhex(mask)));
s.push_str(" \"lanes\": 32,\n");
s.push_str(&format!(" \"source\": {},\n", jstr(source)));
s.push_str(" \"warps\": [\n");
for (i, o) in outs.iter().enumerate() {
s.push_str(&format!(" {{\"base_nonce\": {}, \"expected\": [\n", bases[i]));
for row in 0..4 {
s.push_str(" ");
s.push_str(&(0..8).map(|c| jhex64(o[row * 8 + c])).collect::<Vec<_>>().join(", "));
s.push_str(if row == 3 { "\n" } else { ",\n" });
}
s.push_str(if i == outs.len() - 1 { " ]}\n" } else { " ]},\n" });
}
s.push_str(" ],\n");
s.push_str(&format!(" \"dataset_head\": [{}],\n", join_jhex(&v.head)));
s.push_str(&format!(" \"dataset_last_index\": {mask},\n"));
s.push_str(&format!(" \"dataset_last\": {},\n", jhex(v.last)));
s.push_str(&format!(
" \"dataset_samples\": [{}]",
v.sample_idx
.iter()
.zip(v.sample_val.iter())
.map(|(i, val)| format!("{{\"index\": {i}, \"value\": {}}}", jhex(*val)))
.collect::<Vec<_>>()
.join(", ")
));
if memhard {
s.push_str(&format!(",\n \"cache_head\": [{}],\n", join_jhex(&v.cache_head)));
s.push_str(&format!(" \"cache_last_line\": [{}],\n", join_jhex(&v.cache_last)));
s.push_str(&format!(" \"cache_fnv1a64\": {}\n", jhex64(v.cache_fnv)));
} else {
s.push('\n');
}
s.push_str("}\n");
s
}
/// A program pack: the files `--export-pack` writes, as (name, text).
pub struct Pack {
pub files: Vec<(String, String)>,
pub bases: Vec<u32>,
pub outs: Vec<[u64; 32]>,
pub vectors: PackVectors,
}
impl Pack {
pub fn write_to(&self, dir: &std::path::Path) -> std::io::Result<()> {
std::fs::create_dir_all(dir)?;
for (name, text) in &self.files {
std::fs::write(dir.join(name), text)?;
}
Ok(())
}
}
/// Build the whole pack for an epoch: the three vector warps from the CPU interpreter, the self-test words,
/// and every source file. The `source` string says where the vectors came from.
pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack {
let p = &epoch.program;
let ds: &DatasetSource = &epoch.dataset;
let mask = ds.mask;
let memhard = ds.memhard().map(|m| &m.params);
let bases = PACK_VECTOR_BASES.to_vec();
let outs: Vec<[u64; 32]> = bases.iter().map(|&b| epoch.hash_warp(b)).collect();
let mut v = PackVectors {
head: (0..16).map(|i| ds.word(i)).collect(),
last: ds.word(mask),
sample_idx: sample_indices(mask),
..Default::default()
};
v.sample_val = v.sample_idx.iter().map(|&i| ds.word(i)).collect();
if let Some(m) = ds.memhard() {
let w = m.cache.words();
v.cache_head = w[..16].to_vec();
v.cache_last = w[w.len() - 16..].to_vec();
v.cache_fnv = m.cache.fnv1a64();
}
let is_mh = memhard.is_some();
let mut files = vec![
("program.json".to_string(), program_json(p, day, ds)),
("vectors.json".to_string(), vectors_json(p, day, ds.log2_words, &bases, &outs, &v, mask, source, is_mh)),
("kernel.cu".to_string(), cuda_kernel(p, memhard)),
("kernel.cl".to_string(), opencl_kernel(p, memhard)),
("program.h".to_string(), program_header(p, day, ds)),
("vectors.h".to_string(), vectors_header(p, &bases, &outs, &v, mask, source, is_mh)),
("program.metal".to_string(), metal_program(p, ds.log2_words, LoadSource::Stored)),
// Header-bound kernels (3 October 2026, bind.rs): new files, the seven above are unchanged.
("program_bound.metal".to_string(), metal_program_bound(p, ds.log2_words)),
("kernel_bound.cu".to_string(), cuda_kernel_bound(p, memhard)),
("kernel_bound.cl".to_string(), opencl_kernel_bound(p, memhard)),
];
if let Some(mp) = memhard {
files.push(("memhard.h".to_string(), cuda_memhard_header(p, mp)));
files.push(("memhard.metal".to_string(), metal_memhard(mp)));
}
Pack { files, bases, outs, vectors: v }
}
/// The dataset mode a pack was written in, from its program.json text (no JSON parser needed).
pub fn pack_mode_from_json(program_json: &str) -> DatasetMode {
if program_json.contains("\"dataset_mode\": \"memory-hard\"") {
DatasetMode::MemoryHard
} else {
DatasetMode::ClosedForm
}
}