ds55 (research class, 8 October 2026): a non-power-of-two dataset for the class v3 program through --dataset-words N (2^28 ..= 2^31, a multiple of 2^16); every load address is the multiply-shift range reduction of spec 01 section 1.13.3, idx = (src * N) >> 32 in 64 bits, in the CPU verifier, the CUDA, OpenCL and Metal texts and the self-test vectors; the era window composes in the source space; program.json records dataset.words, bytes, items, mode and the mapping text; the power-of-two path is byte for byte what it was and a power of two through --dataset-words is the --dataset-log2 path; known-failed tests first (tests/ds55.rs)

No consensus object moves. The cache stays 2^26 words; the item index stays 32-bit.

Suite on igneum-build-1 (cargo test --release, the exact sources of this commit):
build-remote: RESULT rc=0 secs=197 compiles=39 class=ok; 124 passed, 0 failed, 7 ignored (ds55: 5 of 5).
The known-failed run before the feature: RESULT rc=101 class=compile-error (tests/ds55.rs against the old crate).

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-labs 2026-10-08 13:48:13 +00:00
parent 3b3c56b245
commit 905c10b836
4 changed files with 671 additions and 67 deletions

View file

@ -16,16 +16,30 @@ use crate::memhard::{
CHACHA_ROUNDS, CHACHA_SIGMA, HOT_TAG, ITEM_ROUNDS,
};
use crate::seed::SplitMix64;
use crate::verify::{window, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2, FOLD_MUL, FOLD_ROT};
use crate::verify::{window, window32, DatasetGeom, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2, FOLD_MUL, FOLD_ROT};
/// The index expression of a dataset load (era layout, `docs/plans/era-layout.md` section 1.3). For every class
/// without an era it is the lottery hash's `rN & MASK`; for an era program it is the one form
/// `((rotl_imm(rN * M, R) & WM) | OFF) & MASK` with the site's window constants at the pack's dataset size.
fn load_index_expr(dialect: CoreDialect, era: Option<&EraParams>, ins: &Instr, a: &str, dataset_log2: u32) -> String {
/// Under the multiply-shift geometry (research class ds55, `--dataset-words`) the AND is the dialect's high
/// multiply by the word count: `mulhi(rN, N)` and `mulhi(((rotl_imm(rN * M, R) & WM32) | OFF32), N)` with the
/// window in the source space (`verify::window32`).
fn load_index_expr(dialect: CoreDialect, era: Option<&EraParams>, ins: &Instr, a: &str, geom: DatasetGeom) -> String {
let mask_name = match dialect {
CoreDialect::Metal => "MASK",
_ => "mask",
};
if geom.mulshift {
let (mulhi, words) = mulhi_name(dialect);
return match era {
None => format!("{mulhi}({a}, {words})"),
Some(e) => {
let (wm, off) = window32(ins, geom.log2);
format!("{mulhi}(((rotl_imm({a} * {}, {}u) & {}) | {}), {words})", hex(e.stride_mul), e.stride_rot, hex(wm), hex(off))
}
};
}
let dataset_log2 = geom.log2;
match era {
None => format!("{a} & {mask_name}"),
Some(e) => {
@ -35,6 +49,47 @@ fn load_index_expr(dialect: CoreDialect, era: Option<&EraParams>, ins: &Instr, a
}
}
/// The dialect's high 32-bit multiply and the name of the word-count constant (the multiply-shift geometry).
fn mulhi_name(dialect: CoreDialect) -> (&'static str, &'static str) {
match dialect {
CoreDialect::Metal => ("mulhi", "DS_WORDS"),
CoreDialect::Cuda => ("__umulhi", "IGNEUM_DS_WORDS"),
CoreDialect::OpenCl => ("mul_hi", "IGNEUM_DS_WORDS"),
}
}
/// The word-count lines of a kernel under the multiply-shift geometry (empty under the mask path, so every pinned
/// pack keeps its text): the define the load expressions read, and the comment that says what moved.
fn ds_words_lines(dialect: CoreDialect, geom: DatasetGeom) -> String {
if !geom.mulshift {
return String::new();
}
let (mulhi, words) = mulhi_name(dialect);
let mut s = String::new();
s.push_str(&format!(
"// Research class ds55 (8 October 2026, NOT the lottery hash): the dataset holds {} words ({} items, {} bytes),\n",
geom.words,
geom.items(),
geom.bytes()
));
s.push_str("// not a power of two. Every load address is the multiply-shift range reduction of spec 01 section 1.13.3,\n");
s.push_str(&format!("// idx = (src * {words}) >> 32 in 64 bits ({mulhi}), in place of src & mask; the mask argument is not read by a load.\n"));
s.push_str(&format!("#define {words} {}\n", hex(geom.words as u32)));
s
}
/// The warp-coalesced load's base expression (lever b, `wload`): lane 0's register range-reduced and aligned
/// down to 32 words.
fn wload_base_expr(dialect: CoreDialect, bcast: &str, geom: DatasetGeom) -> String {
if geom.mulshift {
let (mulhi, words) = mulhi_name(dialect);
format!("({mulhi}({bcast}, {words}) & ~31u) + lane")
} else {
let wmask = if dialect == CoreDialect::Metal { "WMASK" } else { "wmask" };
format!("({bcast} & {wmask}) + lane")
}
}
/// The era lines of program.h (empty without an era).
fn era_header_lines(p: &Program) -> String {
let Some(e) = p.class.era else { return String::new() };
@ -861,23 +916,34 @@ const DS_ELEM_BODY: &str = " x *= 0x9E3779B1u; x ^= x >> 15;\n x += d1;\n
/// The Metal hash kernel (`generateMSL`, program.metal).
pub fn metal_program(p: &Program, dataset_log2: u32, source: LoadSource) -> String {
metal_program_impl(p, dataset_log2, source, false)
metal_program_impl(p, DatasetGeom::pow2(dataset_log2), source, false)
}
/// [`metal_program`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`).
pub fn metal_program_geom(p: &Program, geom: DatasetGeom, source: LoadSource) -> String {
metal_program_impl(p, geom, source, false)
}
/// The header-bound Metal kernel (`program_bound.metal`, serve mode of proto-metal): `igneum_hash_bound` reads its
/// init words `I` from `constant uint* initw [[buffer(3)]]` (`bind::block_init_words`) instead of `SEEDW`. Same
/// instruction text as `igneum_hash`. Stored dataset only.
pub fn metal_program_bound(p: &Program, dataset_log2: u32) -> String {
metal_program_impl(p, dataset_log2, LoadSource::Stored, true)
metal_program_impl(p, DatasetGeom::pow2(dataset_log2), LoadSource::Stored, true)
}
fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: bool) -> String {
let mask = mask_for(dataset_log2);
/// [`metal_program_bound`] at a dataset geometry.
pub fn metal_program_bound_geom(p: &Program, geom: DatasetGeom) -> String {
metal_program_impl(p, geom, LoadSource::Stored, true)
}
fn metal_program_impl(p: &Program, geom: DatasetGeom, source: LoadSource, bound: bool) -> String {
let mask = geom.mask();
let mut s = String::with_capacity(5000);
s.push_str("#include <metal_stdlib>\n");
s.push_str("using namespace metal;\n");
s.push('\n');
s.push_str(&format!("#define MASK {}\n", hex(mask)));
s.push_str(&ds_words_lines(CoreDialect::Metal, geom));
s.push_str(&hot_define(p));
s.push_str(&format!("constant uint SEEDW[8] = {{ {} }};\n", join_hex(&p.seed)));
s.push('\n');
@ -893,7 +959,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound:
s.push_str(" uint x = i ^ d0;\n");
s.push_str(DS_ELEM_BODY);
s.push('\n');
if p.has_wide() {
if p.has_wide() && !geom.mulshift {
s.push_str("#define WMASK (MASK & ~31u)\n\n");
}
let mut buffer0 = "device const uint* dataset [[buffer(0)]]";
@ -945,9 +1011,9 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound:
let era = p.class.era;
let word_index = |a: &str, wide: bool, ins: &Instr| -> String {
if wide {
format!("(simd_broadcast({a}, 0) & WMASK) + lane")
wload_base_expr(CoreDialect::Metal, &format!("simd_broadcast({a}, 0)"), geom)
} else {
load_index_expr(CoreDialect::Metal, era.as_ref(), ins, a, dataset_log2)
load_index_expr(CoreDialect::Metal, era.as_ref(), ins, a, geom)
}
};
let fetch = |idx: String| -> String {
@ -1028,7 +1094,7 @@ fn init_line(p: &Program, u: &str, i: usize) -> String {
}
/// The instruction lines of the CUDA hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`).
fn cuda_instr_lines(p: &Program, dataset_log2: u32) -> String {
fn cuda_instr_lines(p: &Program, geom: DatasetGeom) -> String {
let mut s = String::with_capacity(6000);
let era = p.class.era;
for (k, ins) in p.instrs.iter().enumerate() {
@ -1053,10 +1119,10 @@ fn cuda_instr_lines(p: &Program, dataset_log2: u32) -> String {
Op::Mad => format!("{d} = {a} * {b} + {d};"),
Op::Shfl => format!("{d} = {d} ^ __shfl_xor_sync(0xffffffffu, {a}, {});", ins.mask),
Op::Load if load_width(ins) > 1 => {
wide_load_stmt(CoreDialect::Cuda, &d, &load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None)
wide_load_stmt(CoreDialect::Cuda, &d, &load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, geom), ins.width, WideSource::Stored, None)
}
Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2)),
Op::WLoad => format!("{d} = {d} ^ ds[(__shfl_sync(0xffffffffu, {a}, 0) & wmask) + lane];"),
Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, geom)),
Op::WLoad => format!("{d} = {d} ^ ds[{}];", wload_base_expr(CoreDialect::Cuda, &format!("__shfl_sync(0xffffffffu, {a}, 0)"), geom)),
Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a, p.class.scratch_slot_mask()),
Op::Hot => hot_stmt(CoreDialect::Cuda, &d, &a),
};
@ -1073,6 +1139,11 @@ pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String {
/// [`cuda_kernel`] at a dataset size (an era program's window constants are literals of the pack's size; every
/// other class ignores it).
pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String {
cuda_kernel_geom(p, memhard, DatasetGeom::pow2(dataset_log2))
}
/// [`cuda_kernel_at`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`).
pub fn cuda_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String {
let layout = p.class.layout();
let mut s = String::with_capacity(9000);
s.push_str(&generated_by(&p.seed_string));
@ -1087,6 +1158,7 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3
s.push_str("#include \"memhard.h\"\n");
}
s.push('\n');
s.push_str(&ds_words_lines(CoreDialect::Cuda, geom));
s.push_str(&hot_define(p));
s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n");
s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n");
@ -1164,13 +1236,16 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3
s.push_str(" uint32_t nonce = baseNonce + gid;\n");
s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n");
if p.has_wide() {
s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n");
s.push_str(" uint32_t lane = threadIdx.x & 31u;\n");
if !geom.mulshift {
s.push_str(" uint32_t wmask = mask & ~31u;\n");
}
}
for i in 0..8 {
s.push_str(&init_line(p, "uint32_t", i));
}
s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n"));
s.push_str(&cuda_instr_lines(p, dataset_log2));
s.push_str(&cuda_instr_lines(p, geom));
s.push_str(&shadow_block(p, CoreDialect::Cuda));
s.push_str(" }\n");
s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
@ -1272,6 +1347,11 @@ pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String {
/// [`cuda_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]).
pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String {
cuda_kernel_bound_geom(p, memhard, DatasetGeom::pow2(dataset_log2))
}
/// [`cuda_kernel_bound_at`] at a dataset geometry.
pub fn cuda_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String {
let mut s = String::with_capacity(9000);
s.push_str(&generated_by(&p.seed_string));
s.push_str(
@ -1288,6 +1368,7 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo
s.push_str("#include <cstdint>\n");
s.push_str("#include \"program.h\"\n");
s.push('\n');
s.push_str(&ds_words_lines(CoreDialect::Cuda, geom));
s.push_str("struct IgneumInitWords { uint32_t w[8]; };\n");
s.push('\n');
s.push_str(&hot_define(p));
@ -1313,7 +1394,10 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo
s.push_str(" uint32_t nonce = baseNonce + gid;\n");
s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n");
if p.has_wide() {
s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n");
s.push_str(" uint32_t lane = threadIdx.x & 31u;\n");
if !geom.mulshift {
s.push_str(" uint32_t wmask = mask & ~31u;\n");
}
}
for i in 0..8 {
s.push_str(&format!(
@ -1323,7 +1407,7 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo
));
}
s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n"));
s.push_str(&cuda_instr_lines(p, dataset_log2));
s.push_str(&cuda_instr_lines(p, geom));
s.push_str(&shadow_block(p, CoreDialect::Cuda));
s.push_str(" }\n");
s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
@ -1369,7 +1453,7 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo
}
/// The instruction lines of the OpenCL hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`).
fn opencl_instr_lines(p: &Program, dataset_log2: u32) -> String {
fn opencl_instr_lines(p: &Program, geom: DatasetGeom) -> String {
let mut s = String::with_capacity(6000);
let era = p.class.era;
for (k, ins) in p.instrs.iter().enumerate() {
@ -1393,10 +1477,10 @@ fn opencl_instr_lines(p: &Program, dataset_log2: u32) -> String {
Op::Mad => format!("{d} = {a} * {b} + {d};"),
Op::Shfl => format!("{{ uint t_; IGNEUM_SHFL_XOR(t_, {a}, {}u); {d} = {d} ^ t_; }}", ins.mask),
Op::Load if load_width(ins) > 1 => {
wide_load_stmt(CoreDialect::OpenCl, &d, &load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None)
wide_load_stmt(CoreDialect::OpenCl, &d, &load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, geom), ins.width, WideSource::Stored, None)
}
Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2)),
Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[(t_ & wmask) + lane]; }}"),
Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, geom)),
Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[{}]; }}", wload_base_expr(CoreDialect::OpenCl, "t_", geom)),
Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a, p.class.scratch_slot_mask()),
Op::Hot => hot_stmt(CoreDialect::OpenCl, &d, &a),
};
@ -1414,7 +1498,12 @@ pub fn opencl_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String {
/// [`opencl_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]).
pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String {
let mut s = opencl_kernel_at(p, memhard, dataset_log2);
opencl_kernel_bound_geom(p, memhard, DatasetGeom::pow2(dataset_log2))
}
/// [`opencl_kernel_bound_at`] at a dataset geometry.
pub fn opencl_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String {
let mut s = opencl_kernel_geom(p, memhard, geom);
s.push('\n');
s.push_str(
"// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.\n",
@ -1453,7 +1542,10 @@ pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_
s.push_str("#endif\n");
}
if p.has_wide() {
s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n");
s.push_str(" uint lane = lid & 31u;\n");
if !geom.mulshift {
s.push_str(" uint wmask = mask & ~31u;\n");
}
}
for i in 0..8 {
s.push_str(&format!(
@ -1463,7 +1555,7 @@ pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_
));
}
s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n"));
s.push_str(&opencl_instr_lines(p, dataset_log2));
s.push_str(&opencl_instr_lines(p, geom));
s.push_str(&shadow_block(p, CoreDialect::OpenCl));
s.push_str(" }\n");
s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
@ -1483,6 +1575,11 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String {
/// [`opencl_kernel`] at a dataset size (see [`cuda_kernel_at`]).
pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String {
opencl_kernel_geom(p, memhard, DatasetGeom::pow2(dataset_log2))
}
/// [`opencl_kernel_at`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`).
pub fn opencl_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String {
let layout = p.class.layout();
let mut s = String::with_capacity(14000);
s.push_str(&generated_by(&p.seed_string));
@ -1498,6 +1595,7 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2:
s.push_str("// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.\n");
s.push_str("#ifndef IGNEUM_GROUP\n#define IGNEUM_GROUP 32\n#endif\n");
s.push_str("#ifndef IGNEUM_EXCHANGE\n#define IGNEUM_EXCHANGE 0\n#endif\n");
s.push_str(&ds_words_lines(CoreDialect::OpenCl, geom));
s.push_str("#ifdef __OPENCL_VERSION__\n");
s.push_str("#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))\n");
s.push_str("#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]\n");
@ -1624,13 +1722,16 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2:
s.push_str("#endif\n");
}
if p.has_wide() {
s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n");
s.push_str(" uint lane = lid & 31u;\n");
if !geom.mulshift {
s.push_str(" uint wmask = mask & ~31u;\n");
}
}
for i in 0..8 {
s.push_str(&init_line(p, "uint", i));
}
s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n"));
s.push_str(&opencl_instr_lines(p, dataset_log2));
s.push_str(&opencl_instr_lines(p, geom));
s.push_str(&shadow_block(p, CoreDialect::OpenCl));
s.push_str(" }\n");
s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
@ -1667,7 +1768,8 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String {
let key = &ds.key;
let dataset_log2 = ds.log2_words;
let memhard = ds.memhard().map(|m| &m.params);
let mask = mask_for(dataset_log2);
let geom = ds.geom;
let mask = geom.mask();
let mut s = String::with_capacity(2600);
s.push_str(&generated_by(&p.seed_string));
s.push_str("// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.\n");
@ -1685,8 +1787,20 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String {
s.push_str(&format!("#define IGNEUM_DAY_BYTES_HEX {}\n", jstr(&hex_bytes(&ds.key_bytes))));
s.push_str(&format!("#define IGNEUM_DAY0 {}\n", hex(key[0])));
s.push_str(&format!("#define IGNEUM_DAY1 {}\n", hex(key[1])));
s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n"));
s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask)));
if geom.mulshift {
s.push_str(&format!("// Research class ds55 (8 October 2026): a dataset of {} words, not a power of two. IGNEUM_DATASET_LOG2 is floor(log2(words));\n", geom.words));
s.push_str("// the host allocates IGNEUM_DATASET_WORDS words and builds IGNEUM_DATASET_ITEMS items; IGNEUM_MASK is the last word index (the\n");
s.push_str("// self-test reads dataset[IGNEUM_MASK]) and is never ANDed: every load is idx = (src * IGNEUM_DATASET_WORDS) >> 32 (spec 01 section 1.13.3).\n");
s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n"));
s.push_str(&format!("#define IGNEUM_DATASET_WORDS {}u\n", geom.words));
s.push_str(&format!("#define IGNEUM_DATASET_ITEMS {}u\n", geom.items()));
s.push_str(&format!("#define IGNEUM_DATASET_BYTES {}ull\n", geom.bytes()));
s.push_str("#define IGNEUM_DATASET_MULSHIFT 1\n");
s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask)));
} else {
s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n"));
s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask)));
}
s.push_str("#define IGNEUM_LANES 32\n");
s.push_str(&format!("#define IGNEUM_ITERATIONS {ITERATIONS}\n"));
s.push_str(&format!("#define IGNEUM_INSTR_COUNT {INSTR_COUNT}\n"));
@ -1820,6 +1934,13 @@ pub fn sample_indices(mask: u32) -> Vec<u32> {
(0..64).map(|_| (sr.next() as u32) & mask).collect()
}
/// [`sample_indices`] at a dataset geometry: the same 64 draws through the geometry's range reduction, so the
/// mask path is [`sample_indices`] exactly and the multiply-shift path samples by the mapping the loads use.
pub fn sample_indices_geom(geom: DatasetGeom) -> Vec<u32> {
let mut sr = SplitMix64::new(0x6d68_7361_6d70_6c65);
(0..64).map(|_| geom.reduce(sr.next() as u32)).collect()
}
/// vectors.h (`generateVectorsHeader`).
pub fn vectors_header(
p: &Program,
@ -1903,7 +2024,8 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String {
let key = &ds.key;
let dataset_log2 = ds.log2_words;
let memhard = ds.memhard().map(|m| &m.params);
let mask = mask_for(dataset_log2);
let geom = ds.geom;
let mask = geom.mask();
let mut s = String::with_capacity(14000);
s.push_str("{\n");
s.push_str(" \"format\": \"igneum-program-pack-3\",\n");
@ -1979,7 +2101,11 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String {
s.push_str(&format!(" \"stride_mul\": {},\n", jhex(e.stride_mul)));
s.push_str(&format!(" \"stride_rot\": {},\n", e.stride_rot));
s.push_str(&format!(" \"interleave\": [{}, {}, {}, {}],\n", e.pos[0], e.pos[1], e.pos[2], e.pos[3]));
s.push_str(" \"address\": \"y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words\",\n");
if geom.mulshift {
s.push_str(" \"address\": \"y = rotl(src * stride_mul, stride_rot); D = floor(log2(words)); k = min(win, D - 26); v = (y & (0xffffffff >> k)) | ((off & (2^k - 1)) << (32 - k)); idx = (v * words) >> 32 in 64 bits (the window in the source space, then the multiply-shift of spec 01 section 1.13.3); a wide load aligns idx down to W words\",\n");
} else {
s.push_str(" \"address\": \"y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words\",\n");
}
s.push_str(" \"windows\": \"per instruction, after the width roll: win = below(3), off = low32(next()) & (2^win - 1); used on a load slot (the instruction's win and off fields)\",\n");
s.push_str(" \"dataset_word\": \"dataset[w] = item(t(w))[j(w)]: j(w) gathers the bits of w at the interleave positions, t(w) is w with those bits removed\",\n");
s.push_str(" \"program_id_suffix\": \"'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]\"\n");
@ -2024,17 +2150,34 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String {
s.push_str(
" \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n",
);
s.push_str(" \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n");
s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\"");
if geom.mulshift {
s.push_str(" \"load\": \"dst = dst ^ dataset[(src * dataset.words) >> 32]\",\n");
s.push_str(" \"wload\": \"base = ((src of lane 0 * dataset.words) >> 32) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\"");
} else {
s.push_str(" \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n");
s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\"");
}
if p.has_hot() {
s.push_str(",\n \"hot\": \"dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)\"");
}
s.push('\n');
s.push_str(" },\n");
s.push_str(" \"dataset\": {\n");
s.push_str(&format!(" \"log2_words\": {dataset_log2},\n"));
s.push_str(&format!(" \"bytes\": {},\n", 1u64 << (dataset_log2 as u64 + 2)));
s.push_str(&format!(" \"mask\": {},\n", jhex(mask)));
if geom.mulshift {
s.push_str(&format!(" \"log2_words\": {dataset_log2},\n"));
s.push_str(" \"log2_words_note\": \"floor(log2(words)): the dataset is not a power of two (research class ds55, 8 October 2026); allocate words, build items\",\n");
s.push_str(&format!(" \"words\": {},\n", geom.words));
s.push_str(&format!(" \"bytes\": {},\n", geom.bytes()));
s.push_str(&format!(" \"items\": {},\n", geom.items()));
s.push_str(" \"mapping\": \"mulshift\",\n");
s.push_str(" \"index\": \"idx = (src * words) >> 32 computed in 64 bits (spec 01 section 1.13.3, the multiply-shift range reduction; uniform to within 2^-32, branch-free, integer only), in place of src & mask; the item index is idx >> 4 under the linear layout and t(idx) under an era layout, below items\",\n");
s.push_str(&format!(" \"mask\": {},\n", jhex(mask)));
s.push_str(" \"mask_note\": \"the last word index, words - 1; never ANDed under the multiply-shift\",\n");
} else {
s.push_str(&format!(" \"log2_words\": {dataset_log2},\n"));
s.push_str(&format!(" \"bytes\": {},\n", 1u64 << (dataset_log2 as u64 + 2)));
s.push_str(&format!(" \"mask\": {},\n", jhex(mask)));
}
s.push_str(&format!(" \"day\": {},\n", jstr(day)));
s.push_str(&format!(" \"day_bytes\": {},\n", jstr(&hex_bytes(&ds.key_bytes))));
s.push_str(" \"day_words_from\": \"seed_words_from_bytes(day_bytes)\",\n");
@ -2178,12 +2321,36 @@ pub fn vectors_json(
source: &str,
memhard: bool,
) -> String {
let geom = DatasetGeom::pow2(dataset_log2);
debug_assert_eq!(mask, geom.mask());
vectors_json_geom(p, day, geom, bases, outs, v, source, memhard)
}
/// [`vectors_json`] at a dataset geometry: under the multiply-shift the file also carries `dataset_words` and
/// `dataset_mapping`, and `dataset_last_index` is `words - 1`.
#[allow(clippy::too_many_arguments)]
pub fn vectors_json_geom(
p: &Program,
day: &str,
geom: DatasetGeom,
bases: &[u32],
outs: &[[u64; 32]],
v: &PackVectors,
source: &str,
memhard: bool,
) -> String {
let dataset_log2 = geom.log2;
let mask = geom.mask();
let mut s = String::with_capacity(6500);
s.push_str("{\n");
s.push_str(&format!(" \"seed\": {},\n", jstr(&p.seed_string)));
s.push_str(&format!(" \"day\": {},\n", jstr(day)));
s.push_str(&format!(" \"dataset_mode\": {},\n", jstr(if memhard { "memory-hard" } else { "closed-form" })));
s.push_str(&format!(" \"dataset_log2_words\": {dataset_log2},\n"));
if geom.mulshift {
s.push_str(&format!(" \"dataset_words\": {},\n", geom.words));
s.push_str(" \"dataset_mapping\": \"mulshift\",\n");
}
s.push_str(&format!(" \"mask\": {},\n", jhex(mask)));
s.push_str(" \"lanes\": 32,\n");
s.push_str(&format!(" \"source\": {},\n", jstr(source)));
@ -2255,15 +2422,17 @@ impl Pack {
pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack {
let p = &epoch.program;
let ds: &DatasetSource = &epoch.dataset;
let mask = ds.mask;
let geom = ds.geom;
let mask = geom.mask();
let memhard = ds.memhard().map(|m| &m.params);
let bases = PACK_VECTOR_BASES.to_vec();
let outs: Vec<[u64; 32]> = bases.iter().map(|&b| epoch.hash_warp(b)).collect();
// the self-test words under the program's layout (era layout; linear for every other class)
// the self-test words under the program's layout (era layout; linear for every other class) and the
// geometry's range reduction (the sampled indices go through the mapping the loads use)
let mut v = PackVectors {
head: (0..16).map(|i| epoch.dataset_word(i)).collect(),
last: epoch.dataset_word(mask),
sample_idx: sample_indices(mask),
last: epoch.dataset_word(geom.last_index()),
sample_idx: sample_indices_geom(geom),
..Default::default()
};
v.sample_val = v.sample_idx.iter().map(|&i| epoch.dataset_word(i)).collect();
@ -2284,16 +2453,16 @@ pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack {
let is_mh = memhard.is_some();
let mut files = vec![
("program.json".to_string(), program_json(p, day, ds)),
("vectors.json".to_string(), vectors_json(p, day, ds.log2_words, &bases, &outs, &v, mask, source, is_mh)),
("kernel.cu".to_string(), cuda_kernel_at(p, memhard, ds.log2_words)),
("kernel.cl".to_string(), opencl_kernel_at(p, memhard, ds.log2_words)),
("vectors.json".to_string(), vectors_json_geom(p, day, geom, &bases, &outs, &v, source, is_mh)),
("kernel.cu".to_string(), cuda_kernel_geom(p, memhard, geom)),
("kernel.cl".to_string(), opencl_kernel_geom(p, memhard, geom)),
("program.h".to_string(), program_header(p, day, ds)),
("vectors.h".to_string(), vectors_header(p, &bases, &outs, &v, mask, source, is_mh)),
("program.metal".to_string(), metal_program(p, ds.log2_words, LoadSource::Stored)),
("program.metal".to_string(), metal_program_geom(p, geom, LoadSource::Stored)),
// Header-bound kernels (3 October 2026, bind.rs): new files, the seven above are unchanged.
("program_bound.metal".to_string(), metal_program_bound(p, ds.log2_words)),
("kernel_bound.cu".to_string(), cuda_kernel_bound_at(p, memhard, ds.log2_words)),
("kernel_bound.cl".to_string(), opencl_kernel_bound_at(p, memhard, ds.log2_words)),
("program_bound.metal".to_string(), metal_program_bound_geom(p, geom)),
("kernel_bound.cu".to_string(), cuda_kernel_bound_geom(p, memhard, geom)),
("kernel_bound.cl".to_string(), opencl_kernel_bound_geom(p, memhard, geom)),
];
if let Some(mp) = memhard {
files.push(("memhard.h".to_string(), cuda_memhard_header(p, mp)));

View file

@ -1,7 +1,7 @@
//! igneum-pow CLI.
//!
//! igneum-pow bench --seed <s> [--day <d>] [--closed-form] [--dataset-log2 28] [--warps 20]
//! igneum-pow export --seed <s> --out <dir> [--day <d>] [--closed-form] [--dataset-log2 28] [--epoch-hex <64 hex> --day-hex <hex>]
//! igneum-pow export --seed <s> --out <dir> [--day <d>] [--closed-form] [--dataset-log2 28 | --dataset-words N] [--epoch-hex <64 hex> --day-hex <hex>]
//! igneum-pow hash --seed <s> --nonce <n> [--day <d>] [--closed-form] [--dataset-log2 28]
//! igneum-pow hash-bound --seed <s> --prehash <64 hex> --nonce <u64> [--day <d>] [--closed-form] [--dataset-log2 28]
//! [--epoch-hex <64 hex> --day-hex <hex>] byte seeds instead of strings (Epoch::from_seed_bytes)
@ -22,7 +22,7 @@ use igneum_pow::emit::export_pack;
use igneum_pow::generator::{LoadClass, ProgramClass};
use igneum_pow::memhard::{Cache, Shape};
use igneum_pow::seed::day_key;
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2};
use igneum_pow::verify::{DatasetGeom, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2};
use std::time::Instant;
struct Args {
@ -32,6 +32,10 @@ struct Args {
out: Option<String>,
closed_form: bool,
dataset_log2: u32,
/// `--dataset-words N`: the dataset at N words (research class ds55, 8 October 2026): any multiple of 2^16 in
/// 2^28 ..= 2^31; a non-power-of-two takes the multiply-shift range reduction of spec 01 section 1.13.3 in every
/// load, a power of two is the `--dataset-log2` path. Applied after the class's own sizing.
dataset_words: Option<DatasetGeom>,
warps: usize,
nonce: u64,
/// `hash-bound --count N`: N consecutive nonces from --nonce, one epoch build (gate G2, 5 October 2026).
@ -109,7 +113,8 @@ fn usage() -> ! {
\x20 --state <file> class v5 (or any --class ...+state): the window's state stream (IGSD1 file, igneum-day-stream --out), whose leaves key every item\n\
\x20 --shadow-reps N class v4 at a rung of the latency ladder: the shadow block's pass count (0 = the class's own 27; docs/design/latency-ladder.md), with --program-class v4\n\
\x20 --era E era layout over --class: igneum-era-test/<n> or <n>:<64 hex> (the 32-byte era seed E_n)\n\
\x20 --era-widths 4[,16,32,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it; 32 only with the w32 class)"
\x20 --era-widths 4[,16,32,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it; 32 only with the w32 class)\n\
\x20 --dataset-words N research class ds55: the dataset at N words (a multiple of 65,536 in 2^28 ..= 2^31; 1476395008 = 5.5 GiB); a non-power-of-two uses idx = (src * N) >> 32 in every load (spec 01 section 1.13.3), a power of two is --dataset-log2"
);
std::process::exit(2)
}
@ -123,6 +128,7 @@ fn parse() -> Args {
closed_form: false,
state: None,
dataset_log2: DEFAULT_DATASET_LOG2,
dataset_words: None,
warps: 20,
nonce: 0,
count: 1,
@ -147,6 +153,13 @@ fn parse() -> Args {
"--out" => a.out = Some(val()),
"--closed-form" => a.closed_form = true,
"--dataset-log2" => a.dataset_log2 = val().parse().unwrap_or_else(|_| usage()),
"--dataset-words" => {
let n: u64 = val().parse().unwrap_or_else(|_| usage());
a.dataset_words = Some(DatasetGeom::words(n).unwrap_or_else(|err| {
eprintln!("{err}");
std::process::exit(2)
}));
}
"--warps" => a.warps = val().parse().unwrap_or_else(|_| usage()),
"--nonce" => a.nonce = val().parse().unwrap_or_else(|_| usage()),
"--count" => a.count = val().parse().unwrap_or_else(|_| usage()),
@ -223,6 +236,15 @@ fn main() {
fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) {
let (mut e, label) = epoch_of_class(a, mode);
stamp_era(&mut e, a);
// research class ds55: the dataset at the word count of --dataset-words, the cache and the items unchanged; a
// state class sizes its leaves by the power-of-two count and is refused here
if let Some(geom) = a.dataset_words {
if e.program.class.state {
eprintln!("--dataset-words is not supported with a state class ({}): the leaves are sized by --dataset-log2", e.program.class.name());
std::process::exit(2);
}
e.dataset = e.dataset.with_geom(geom);
}
// class v5: the leaves of --state, built for the dataset's size; a state class without --state is refused here
// rather than at the first derivation
if e.program.class.state {
@ -382,10 +404,10 @@ fn export(a: &Args, mode: DatasetMode) {
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
println!("igneum-pow export {out}");
println!(
"seed \"{}\", day \"{}\", dataset 2^{} words ({}), generator v{} attempt {} program id {:016x}, loads/hash {}; epoch built in {build_ms:.1} ms",
"seed \"{}\", day \"{}\", dataset {} ({}), generator v{} attempt {} program id {:016x}, loads/hash {}; epoch built in {build_ms:.1} ms",
e.program.seed_string,
day_label,
e.dataset.log2_words,
e.dataset.geom.describe(),
e.dataset.mode().name(),
e.program.generator,
e.program.attempt,
@ -393,6 +415,10 @@ fn export(a: &Args, mode: DatasetMode) {
e.program.loads_per_hash()
);
println!("op mix: {}; class {}, {} bytes/hash, widths (1,4,16 words) {:?}", e.program.op_mix(), e.program.class.name(), e.program.bytes_per_hash(), e.program.width_counts());
println!("seed words {}", e.program.seed.iter().map(|w| format!("0x{w:08x}")).collect::<Vec<_>>().join(" "));
if e.dataset.geom.mulshift {
println!("dataset mapping: multiply-shift, idx = (src * {}) >> 32; {} items, {} bytes (research class ds55)", e.dataset.geom.words, e.dataset.geom.items(), e.dataset.geom.bytes());
}
let source = format!("igneum-pow (Rust) CPU interpreter, generator v{}, {} dataset", e.program.generator, e.dataset.mode().name());
let pack = export_pack(&e, &day_label, &source);
let dir = std::path::Path::new(&out);

View file

@ -31,6 +31,122 @@ pub fn window(ins: &Instr, mask: u32, log2: u32) -> (u32, u32) {
(wm, off)
}
/// The window of a load site in the 32-bit SOURCE space (the multiply-shift mapping, research class ds55,
/// 8 October 2026): `k = min(win, log2 - 26)` as [`window`] with `log2 = floor(log2(N))`, and `(window mask,
/// offset)` such that `v = (y & window mask) | offset` lies in the site's aligned window of `2^(32 - k)` source
/// values; `idx = (v * N) >> 32` then lands in a contiguous run of about `N / 2^k` words, the site's window of the
/// dataset.
#[inline(always)]
pub fn window32(ins: &Instr, log2: u32) -> (u32, u32) {
let k = (ins.win as u32).min(log2.saturating_sub(26));
let wm = u32::MAX >> k;
let off = ((((ins.off as u32) & ((1u32 << k) - 1)) as u64) << (32 - k)) as u32;
(wm, off)
}
/// The dataset's geometry: `2^log2` words under the lottery hash's `src AND MASK` (`mulshift` false), or `words`
/// words, any multiple of 2^16 in `2^28 ..= 2^31`, under the multiply-shift range reduction of spec 01 section
/// 1.13.3 (`mulshift` true: `idx = (src * words) >> 32` in 64 bits; research class ds55, 8 October 2026, no
/// consensus object moves). A power of two given as a word count takes the mask path, so `--dataset-words 2^28`
/// is `--dataset-log2 28` byte for byte. `log2` is `floor(log2(words))` under the multiply-shift (the era
/// window's floor rule reads it); the item index `words / 16 - 1` stays 32-bit.
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
pub struct DatasetGeom {
pub log2: u32,
pub words: u64,
pub mulshift: bool,
}
impl DatasetGeom {
/// The smallest word count `--dataset-words` takes: 2^28 (1 GiB).
pub const MIN_WORDS: u64 = 1 << 28;
/// The largest: 2^31 (8 GiB; the item index stays 32-bit far beyond, the kernel's word index is 32-bit).
pub const MAX_WORDS: u64 = 1 << 31;
/// The step: 2^16 words, so the era layout's interleave (positions below 16) stays a bijection of the dataset
/// and every wide or warp-coalesced load stays inside it.
pub const WORDS_STEP: u64 = 1 << 16;
/// A dataset of `2^log2` words (4 to 32), the lottery hash's mask path.
pub fn pow2(log2: u32) -> Self {
assert!((4..=32).contains(&log2), "dataset log2 must be in 4..=32");
Self { log2, words: 1u64 << log2, mulshift: false }
}
/// A dataset of `words` words: the mask path for a power of two, the multiply-shift otherwise. Refused
/// outside `MIN_WORDS ..= MAX_WORDS` or off the `WORDS_STEP` grid, with the reason.
pub fn words(words: u64) -> Result<Self, String> {
if !(Self::MIN_WORDS..=Self::MAX_WORDS).contains(&words) {
return Err(format!(
"--dataset-words {words}: the word count must be in 2^28 ..= 2^31 ({} ..= {})",
Self::MIN_WORDS,
Self::MAX_WORDS
));
}
if words % Self::WORDS_STEP != 0 {
return Err(format!("--dataset-words {words}: the word count must be a multiple of 2^16 words (65,536; the era interleave and the wide loads)"));
}
if words.is_power_of_two() {
return Ok(Self::pow2(words.trailing_zeros()));
}
Ok(Self { log2: words.ilog2(), words, mulshift: true })
}
/// The last word index (`words - 1`). Under the mask path it is the AND mask.
pub fn last_index(&self) -> u32 {
(self.words - 1) as u32
}
/// The AND mask of the mask path; under the multiply-shift the last index (recorded in packs as `IGNEUM_MASK`
/// for the host's self-test, never ANDed).
pub fn mask(&self) -> u32 {
self.last_index()
}
pub fn items(&self) -> u64 {
self.words >> 4
}
pub fn bytes(&self) -> u64 {
self.words << 2
}
/// The range reduction of a 32-bit source value to a word index.
#[inline(always)]
pub fn reduce(&self, x: u32) -> u32 {
if self.mulshift {
((x as u64 * self.words) >> 32) as u32
} else {
x & self.last_index()
}
}
/// One line for logs: `2^28 words` or `1476395008 words (5.50 GiB, not a power of two, multiply-shift)`.
pub fn describe(&self) -> String {
if self.mulshift {
format!("{} words ({:.2} GiB, not a power of two, multiply-shift)", self.words, self.bytes() as f64 / (1u64 << 30) as f64)
} else {
format!("2^{} words", self.log2)
}
}
}
/// [`load_index`] at a dataset geometry: the mask path unchanged; under the multiply-shift the plain load is
/// `(x * N) >> 32` and the era form windows the source first ([`window32`]) then reduces.
#[inline(always)]
pub fn load_index_geom(era: Option<&EraParams>, ins: &Instr, x: u32, geom: DatasetGeom) -> u32 {
if !geom.mulshift {
return load_index(era, ins, x, geom.mask(), geom.log2);
}
match era {
None => geom.reduce(x),
Some(e) => {
let (wm, off) = window32(ins, geom.log2);
let y = x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot);
geom.reduce((y & wm) | off)
}
}
}
/// Read-width experiment (5 October 2026): a `load` of `W` words folds every word into `dst`:
/// `x = dst XOR w[0]; for j in 1..W: x = (rotl(x, FOLD_ROT) * FOLD_MUL) XOR w[j]; dst = x`. For `W = 1` this is the
/// lottery hash's `dst XOR dataset[...]`. The fold is state-dependent (the rotate-multiply sits between the words),
@ -184,8 +300,12 @@ pub enum Dataset {
/// A dataset of `2^log2` words plus the construction that fills it.
pub struct DatasetSource {
/// `floor(log2(words))`: the size under the mask path, the era window's floor under the multiply-shift.
pub log2_words: u32,
/// The last word index: the AND mask under the mask path (see [`DatasetGeom::mask`]).
pub mask: u32,
/// The geometry (size and range reduction); `log2_words` and `mask` are its `log2` and `mask()`.
pub geom: DatasetGeom,
/// The day key `K`; `d0, d1 = K[0], K[1]`.
pub key: [u32; 8],
/// The bytes `K` was derived from (`"day/<day>"` for a string day, `bind::day_bytes` on the chain), recorded
@ -219,12 +339,25 @@ impl DatasetSource {
/// `2^shape.cache_log2_words` words on the calling thread.
pub fn from_key_shape(key: [u32; 8], mode: DatasetMode, log2_words: u32, shape: Shape) -> Self {
assert!((4..=32).contains(&log2_words), "dataset log2 must be in 4..=32");
let mask = if log2_words == 32 { u32::MAX } else { (1u32 << log2_words) - 1 };
Self::from_key_geom(key, mode, DatasetGeom::pow2(log2_words), shape)
}
/// [`DatasetSource::from_key_shape`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`).
pub fn from_key_geom(key: [u32; 8], mode: DatasetMode, geom: DatasetGeom, shape: Shape) -> Self {
let dataset = match mode {
DatasetMode::ClosedForm => Dataset::ClosedForm { d0: key[0], d1: key[1] },
DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::with_shape(key, shape)),
};
Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None }
Self { log2_words: geom.log2, mask: geom.mask(), geom, key, key_bytes: Vec::new(), dataset, hot: None }
}
/// This source at another geometry: the cache, key and construction unchanged (an item has the same value at
/// every size), only the size and the range reduction move.
pub fn with_geom(mut self, geom: DatasetGeom) -> Self {
self.log2_words = geom.log2;
self.mask = geom.mask();
self.geom = geom;
self
}
/// This source with the window's state leaves (class v5, `docs/design/class-v5-stored-state.md`): memory-hard mode
@ -246,7 +379,7 @@ impl DatasetSource {
Dataset::MemoryHard(m) => Dataset::MemoryHard(m.refreshed(leaves)),
Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"),
};
Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None }
Self { log2_words: self.log2_words, mask: self.mask, geom: self.geom, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None }
}
/// A copy of this source sharing its cache (and leaves), for a caller that needs an owned source from a shared one.
@ -255,7 +388,7 @@ impl DatasetSource {
Dataset::MemoryHard(m) => Dataset::MemoryHard(crate::memhard::MemhardCpu { params: m.params.clone(), cache: m.cache.clone(), leaves: m.leaves.clone() }),
Dataset::ClosedForm { d0, d1 } => Dataset::ClosedForm { d0: *d0, d1: *d1 },
};
Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None }
Self { log2_words: self.log2_words, mask: self.mask, geom: self.geom, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None }
}
/// The window's state leaves, when the source carries them.
@ -299,8 +432,14 @@ impl DatasetSource {
}
/// `dataset[w & mask]` under a program's layout (era layout). The closed form has no items and ignores it.
/// Under the multiply-shift geometry `w` must already be a word index (below `words`).
pub fn word_at(&self, layout: Layout, w: u32) -> u32 {
let w = w & self.mask;
let w = if self.geom.mulshift {
assert!((w as u64) < self.geom.words, "word {w} outside a dataset of {} words", self.geom.words);
w
} else {
w & self.mask
};
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => dataset_elem(w, *d0, *d1),
Dataset::MemoryHard(m) => m.word_at(layout, w),
@ -375,8 +514,7 @@ pub fn interpret_warp_scratch(
ds: &DatasetSource,
trace: bool,
) -> (WarpResult, Vec<ScratchEvent>) {
let mask = ds.mask;
let log2 = ds.log2_words;
let geom = ds.geom;
let era = program.class.era;
let layout = program.class.layout();
let mut r = [[0u32; LANES]; 8];
@ -406,7 +544,7 @@ pub fn interpret_warp_scratch(
for _ in 0..ITERATIONS {
let sel = r[0];
for ins in &program.instrs {
step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived);
step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived);
if ins.op == Op::Scratch {
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
let (d, a) = (ins.dst as usize, ins.src as usize);
@ -420,7 +558,7 @@ pub fn interpret_warp_scratch(
// iteration's `sel`; it is empty on every class without a shadow, so version 2 and class v3 run nothing here.
for _ in 0..program.shadow_reps() {
for ins in &program.shadow {
step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived);
step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived);
}
}
}
@ -440,8 +578,7 @@ fn step(
ins: &Instr,
r: &mut [[u32; LANES]; 8],
sel: &[u32; LANES],
mask: u32,
log2: u32,
geom: DatasetGeom,
era: Option<&EraParams>,
layout: Layout,
ds: &DatasetSource,
@ -519,7 +656,7 @@ fn step(
}
Op::Load if ins.width == 1 => {
for lane in 0..LANES {
idx[lane] = load_index(era, ins, r[a][lane], mask, log2);
idx[lane] = load_index_geom(era, ins, r[a][lane], geom);
}
*items_derived += ds.fetch(idx, val, layout);
for lane in 0..LANES {
@ -531,7 +668,7 @@ fn step(
let width = ins.width as usize;
let align = !(ins.width as u32 - 1);
for lane in 0..LANES {
idx[lane] = load_index(era, ins, r[a][lane], mask, log2) & align;
idx[lane] = load_index_geom(era, ins, r[a][lane], geom) & align;
}
let mut vals = [[0u32; 16]; LANES];
*items_derived += ds.fetch_wide(idx, width, &mut vals, layout);
@ -551,8 +688,8 @@ fn step(
}
}
Op::WLoad => {
// Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l.
let base = (r[a][0] & mask) & !31;
// Lane 0's register, range-reduced, aligned down to 32 words; lane l reads word base + l.
let base = geom.reduce(r[a][0]) & !31;
for lane in 0..LANES {
idx[lane] = base + lane as u32;
}

272
igneum-pow/tests/ds55.rs Normal file
View file

@ -0,0 +1,272 @@
//! The non-power-of-two dataset of the research class ds55 (8 October 2026): `--dataset-words N` sizes the dataset
//! at N words (N / 16 items) and every load address is the multiply-shift range reduction of spec 01 section 1.13.3,
//! `idx = (src * N) >> 32` in 64 bits, in place of `src & MASK`. No consensus object moves: the power-of-two path is
//! byte for byte what it was, and a power of two given through `--dataset-words` is the `--dataset-log2` path.
//!
//! Known-failed first: every test here was written before the feature and failed to compile against the crate.
use igneum_pow::emit::export_pack;
use igneum_pow::generator::{Op, ProgramClass, LOAD_SLOTS};
use igneum_pow::seed::SplitMix64;
use igneum_pow::verify::{fold_words, load_index_geom, window32, DatasetGeom, Epoch};
use serde_json::Value;
use std::path::PathBuf;
use std::sync::OnceLock;
/// The devnet genesis hash: the epoch seed and the era seed of the pinned class v3 pack mx8-devnet-epoch0.
const EPOCH_HEX: &str = "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07";
/// The day bytes of the pinned pack (`"igneum-day/" || 20730_le64`).
const DAY_HEX: &str = "69676e65756d2d6461792ffa50000000000000";
const PINNED_ID: u64 = 0x73bcbfe8ccf988f1;
const SOURCE: &str = "igneum-pow (Rust) CPU interpreter, generator v3, memory-hard dataset";
/// 5.5 GiB: the genesis schedule's 8 GB-tier reading.
const DS55_WORDS: u64 = 1_476_395_008;
fn pinned_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-mixer/mx8-devnet-epoch0")
}
fn pinned(file: &str) -> String {
let p = pinned_dir().join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
fn day_label() -> String {
format!("bytes:{DAY_HEX}")
}
/// The pinned pack's epoch by the chain's path (the export's `--epoch-hex --day-hex --era-hex --program-class v3`
/// form), at the dataset geometry given. One cache fill per geometry.
fn epoch_at(geom: DatasetGeom) -> Epoch {
let eb = igneum_pow::bind::unhex(EPOCH_HEX).unwrap();
let db = igneum_pow::bind::unhex(DAY_HEX).unwrap();
let label = format!("igneum-epoch/{EPOCH_HEX}/day/{DAY_HEX}");
let program = Epoch::chain_program_shadow(&eb, Some(&eb), ProgramClass::V3, 0, &label);
let dataset = Epoch::chain_dataset_day(&db, ProgramClass::V3, 0, 28).with_geom(geom);
Epoch { program, dataset }
}
fn default_epoch() -> &'static Epoch {
static E: OnceLock<Epoch> = OnceLock::new();
E.get_or_init(|| epoch_at(DatasetGeom::pow2(28)))
}
fn ds55_epoch() -> &'static Epoch {
static E: OnceLock<Epoch> = OnceLock::new();
E.get_or_init(|| epoch_at(DatasetGeom::words(DS55_WORDS).unwrap()))
}
fn ds55_geom() -> DatasetGeom {
DatasetGeom::words(DS55_WORDS).unwrap()
}
/// 1. The default path is byte for byte the pinned pack: the same instructions, the same vectors, every file.
#[test]
fn power_of_two_path_is_byte_identical_to_the_pinned_pack() {
let e = default_epoch();
assert_eq!(e.program.program_id(), PINNED_ID);
assert!(!e.dataset.geom.mulshift);
assert_eq!(e.dataset.geom, DatasetGeom::pow2(28));
let out = export_pack(e, &day_label(), SOURCE);
assert_eq!(out.files.len(), 12, "the pack's twelve files");
for (name, text) in &out.files {
let want = pinned(name);
assert!(text == &want, "mx8-devnet-epoch0/{name} differs from the default path's export");
}
let j: Value = serde_json::from_str(&pinned("program.json")).unwrap();
assert_eq!(j["dataset"]["log2_words"].as_u64().unwrap(), 28);
assert!(j["dataset"].get("words").is_none(), "the power-of-two pack carries no words field");
}
/// 2. A power of two through `--dataset-words` is the `--dataset-log2` path: the same geometry and the same pack.
#[test]
fn dataset_words_power_of_two_equals_dataset_log2() {
let g = DatasetGeom::words(1 << 28).unwrap();
assert_eq!(g, DatasetGeom::pow2(28));
assert!(!g.mulshift);
assert_eq!(g.mask(), 0x0fff_ffff);
for log2 in [28u32, 29, 30, 31] {
assert_eq!(DatasetGeom::words(1u64 << log2).unwrap(), DatasetGeom::pow2(log2), "2^{log2}");
}
let e = epoch_at(g);
let want = export_pack(default_epoch(), &day_label(), SOURCE);
let got = export_pack(&e, &day_label(), SOURCE);
assert_eq!(got.outs, want.outs);
assert_eq!(got.vectors, want.vectors);
for ((n1, t1), (n2, t2)) in got.files.iter().zip(want.files.iter()) {
assert_eq!(n1, n2);
assert!(t1 == t2, "{n1} differs between --dataset-words 2^28 and --dataset-log2 28");
}
// the bounds and the step of the option
assert!(DatasetGeom::words((1 << 28) - 65_536).is_err(), "below 2^28");
assert!(DatasetGeom::words((1 << 31) + 65_536).is_err(), "above 2^31");
assert!(DatasetGeom::words(DS55_WORDS + 1).is_err(), "not a multiple of 2^16 words");
assert!(DatasetGeom::words(DS55_WORDS).is_ok());
assert!(DatasetGeom::words(1 << 31).is_ok());
}
/// 3. The multiply-shift index is in range at every source value and uniform: an exact 2^20 grid lands evenly, and
/// 2^24 SplitMix64 draws fill 64 buckets within 1 percent (at 2^20 draws one bucket's standard deviation is 0.78
/// percent, so 1 percent is not a bound there; at 2^24 it is 0.19 percent and 1 percent is 5 sigma).
#[test]
fn mulshift_index_is_in_range_and_uniform() {
let g = ds55_geom();
assert!(g.mulshift);
assert_eq!(g.words, DS55_WORDS);
assert_eq!(g.log2, 30, "floor(log2(5.5 GiB in words))");
assert_eq!(g.items(), DS55_WORDS / 16);
assert_eq!(g.bytes(), DS55_WORDS * 4);
assert_eq!(g.last_index(), (DS55_WORDS - 1) as u32);
assert_eq!(g.reduce(0), 0);
assert_eq!(g.reduce(u32::MAX), (DS55_WORDS - 1) as u32, "the top source value maps to the last word");
assert_eq!(g.reduce(1 << 31), (DS55_WORDS / 2) as u32, "half the source range is half the dataset");
let bucket = DS55_WORDS / 64;
assert_eq!(bucket * 64, DS55_WORDS);
// the grid: x = i << 12 for i in 0..2^20
let mut hist = [0u64; 64];
for i in 0..(1u32 << 20) {
let idx = g.reduce(i << 12) as u64;
assert!(idx < DS55_WORDS);
hist[(idx / bucket) as usize] += 1;
}
let mean = (1u64 << 20) / 64;
for (b, &h) in hist.iter().enumerate() {
assert!((h as i64 - mean as i64).unsigned_abs() * 100 <= mean, "grid bucket {b}: {h} against {mean}");
}
// the draws
let mut hist = [0u64; 64];
let mut sr = SplitMix64::new(0x6473_3535);
let draws = 1u64 << 24;
let mut max = 0u32;
for _ in 0..draws {
let idx = g.reduce(sr.next() as u32);
assert!((idx as u64) < DS55_WORDS);
max = max.max(idx);
hist[(idx as u64 / bucket) as usize] += 1;
}
assert!(max as u64 >= DS55_WORDS - (1 << 12), "the draws reach the last 4,096 words: {max}");
let mean = draws / 64;
for (b, &h) in hist.iter().enumerate() {
assert!((h as i64 - mean as i64).unsigned_abs() * 100 <= mean, "draw bucket {b}: {h} against {mean} (1 percent)");
}
// against the power-of-two path: the same source through the mask is a different word, the mapping moved
let p = DatasetGeom::pow2(28);
assert_eq!(p.reduce(0xdead_beef), 0xdead_beef & 0x0fff_ffff);
assert_ne!(g.reduce(0xdead_beef), p.reduce(0xdead_beef));
}
/// 4. Fold and address agree between the CPU verifier and the emitted CUDA text: every load line of the bound kernel
/// carries the one multiply-shift era form built from the instruction's own window, and evaluating that text's
/// constants as the kernel would (`__umulhi` = the high 32 bits of the 64-bit product) gives the CPU's index at a
/// thousand source values per site. The fold of a one-word load is the plain xor on both sides.
#[test]
fn cpu_verifier_and_cuda_text_agree_on_fold_and_address() {
let e = ds55_epoch();
let g = e.dataset.geom;
let era = e.program.class.era.expect("the pinned pack is an era program");
let out = export_pack(e, &day_label(), SOURCE);
let cu = &out.files.iter().find(|(n, _)| n == "kernel_bound.cu").unwrap().1;
assert!(cu.contains("#define IGNEUM_DS_WORDS 0x58000000u"), "the kernel names the word count");
let mut sites = 0usize;
let mut sr = SplitMix64::new(0x6373_3535);
for (k, ins) in e.program.instrs.iter().enumerate() {
if ins.op != Op::Load {
continue;
}
sites += 1;
let (wm, off) = window32(ins, g.log2);
let expr = format!(
"__umulhi(((rotl_imm(r{} * 0x{:08x}u, {}u) & 0x{:08x}u) | 0x{:08x}u), IGNEUM_DS_WORDS)",
ins.src, era.stride_mul, era.stride_rot, wm, off
);
let line = format!("r{} = r{} ^ ds[{expr}]; // {k} load", ins.dst, ins.dst);
assert!(cu.lines().any(|l| l.trim() == line), "instruction {k}: the CUDA text lacks\n {line}");
// evaluate the text's constants as the kernel does
for _ in 0..1000 {
let x = sr.next() as u32;
let y = x.wrapping_mul(era.stride_mul).rotate_left(era.stride_rot);
let v = (y & wm) | off;
let kernel = ((v as u64 * DS55_WORDS) >> 32) as u32;
assert_eq!(kernel, load_index_geom(Some(&era), ins, x, g), "instruction {k}, source {x:#010x}");
assert!((kernel as u64) < DS55_WORDS);
// the window: the site's offset bits are the top k bits of the source value
let k_bits = (ins.win as u32).min(g.log2 - 26);
if k_bits > 0 {
assert_eq!(v >> (32 - k_bits), off >> (32 - k_bits), "instruction {k}: the window's top {k_bits} bits");
}
}
assert_eq!(ins.width, 1, "the class v3 pack reads one word a load");
assert_eq!(fold_words(0x1234_5678, &[0x9abc_def0]), 0x1234_5678 ^ 0x9abc_def0);
}
assert_eq!(sites, LOAD_SLOTS);
assert_eq!(cu.lines().filter(|l| l.contains(" & mask]")).count(), 0, "no masked load remains");
assert_eq!(cu.lines().filter(|l| l.contains("__umulhi(((rotl_imm(r")).count(), LOAD_SLOTS);
// the OpenCL and Metal texts carry the same form in their own spelling
let cl = &out.files.iter().find(|(n, _)| n == "kernel_bound.cl").unwrap().1;
assert_eq!(cl.lines().filter(|l| l.contains("mul_hi(((rotl_imm(r")).count(), 2 * LOAD_SLOTS, "two kernels in kernel_bound.cl");
let mt = &out.files.iter().find(|(n, _)| n == "program_bound.metal").unwrap().1;
assert_eq!(mt.lines().filter(|l| l.contains("mulhi(((rotl_imm(r")).count(), LOAD_SLOTS);
assert!(mt.contains("#define DS_WORDS 0x58000000u"));
}
/// The 5.5 GiB pack: the same program id and seed words as the pinned pack (the size does not enter the id), the
/// dataset fields of program.json, the self-test vectors computed by the same mapping (every sampled index below N,
/// every value the verifier's word, the last word at N - 1), and the hashes move from the pinned ones.
#[test]
fn ds55_pack_fields_and_vectors() {
let e = ds55_epoch();
assert_eq!(e.program.program_id(), PINNED_ID);
assert_eq!(e.program.seed[0], 0x667d_0fbd);
assert_eq!(e.program.seed[1], 0x7b8e_5963);
let out = export_pack(e, &day_label(), SOURCE);
assert_eq!(out.files.len(), 12);
let pj = &out.files.iter().find(|(n, _)| n == "program.json").unwrap().1;
let j: Value = serde_json::from_str(pj).expect("program.json is valid JSON");
assert_eq!(j["program_id"].as_str().unwrap(), "0x73bcbfe8ccf988f1");
let d = &j["dataset"];
assert_eq!(d["words"].as_u64().unwrap(), DS55_WORDS);
assert_eq!(d["bytes"].as_u64().unwrap(), DS55_WORDS * 4);
assert_eq!(d["items"].as_u64().unwrap(), DS55_WORDS / 16);
assert_eq!(d["mode"].as_str().unwrap(), "memory-hard");
assert_eq!(d["mapping"].as_str().unwrap(), "mulshift");
assert!(d["index"].as_str().unwrap().contains("(src * words) >> 32"), "the mapping text");
assert_eq!(d["log2_words"].as_u64().unwrap(), 30);
assert_eq!(j["op_semantics"]["load"].as_str().unwrap(), "dst = dst ^ dataset[(src * dataset.words) >> 32]");
assert!(j["era"]["address"].as_str().unwrap().contains(">> 32"));
let vj = &out.files.iter().find(|(n, _)| n == "vectors.json").unwrap().1;
let v: Value = serde_json::from_str(vj).expect("vectors.json is valid JSON");
assert_eq!(v["dataset_words"].as_u64().unwrap(), DS55_WORDS);
assert_eq!(v["dataset_last_index"].as_u64().unwrap(), DS55_WORDS - 1);
assert_eq!(v["dataset_mapping"].as_str().unwrap(), "mulshift");
let last = u32::from_str_radix(v["dataset_last"].as_str().unwrap().trim_start_matches("0x"), 16).unwrap();
assert_eq!(last, e.dataset_word((DS55_WORDS - 1) as u32));
let samples = v["dataset_samples"].as_array().unwrap();
assert_eq!(samples.len(), 64);
let mut above_2_30 = 0;
for s in samples {
let idx = s["index"].as_u64().unwrap();
assert!(idx < DS55_WORDS, "sample index {idx} inside the dataset");
if idx >= 1 << 30 {
above_2_30 += 1;
}
let val = u32::from_str_radix(s["value"].as_str().unwrap().trim_start_matches("0x"), 16).unwrap();
assert_eq!(val, e.dataset_word(idx as u32), "sample {idx}");
}
assert!(above_2_30 >= 10, "the samples reach past 2^30 words ({above_2_30} of 64)");
assert_eq!(out.vectors.sample_idx, igneum_pow::emit::sample_indices_geom(e.dataset.geom));
// the pinned vectors are the power-of-two mapping's; the 5.5 GiB hashes differ
let pinned_v: Value = serde_json::from_str(&pinned("vectors.json")).unwrap();
let pinned_lane0 = u64::from_str_radix(pinned_v["warps"][0]["expected"][0].as_str().unwrap().trim_start_matches("0x"), 16).unwrap();
assert_eq!(default_epoch().hash_warp(0)[0], pinned_lane0);
assert_ne!(out.outs[0][0], pinned_lane0);
// and the verifier's single-nonce path agrees with the warp
assert_eq!(e.hash(4096 + 3), out.outs[1][3]);
// program.h names the geometry for the host
let ph = &out.files.iter().find(|(n, _)| n == "program.h").unwrap().1;
assert!(ph.contains("#define IGNEUM_DATASET_WORDS 1476395008u\n"));
assert!(ph.contains("#define IGNEUM_DATASET_ITEMS 92274688u\n"));
assert!(ph.contains("#define IGNEUM_DATASET_MULSHIFT 1\n"));
assert!(ph.contains("#define IGNEUM_MASK 0x57ffffffu\n"), "IGNEUM_MASK is the last index under mulshift");
let vh = &out.files.iter().find(|(n, _)| n == "vectors.h").unwrap().1;
assert!(vh.contains("static const uint32_t IGNEUM_DS_LAST_INDEX = 1476395007u;\n"));
}