igneum-pow: Rust crate bit-exact with proto-metal (seed, generator, memhard, verifier, emitters)

Standard-library Rust port of the Swift prototype for the rusty-kaspa fork. 23 tests
tie it to proto-cuda/packs: program.json instruction by instruction for three packs,
cache FNV 48c4f5bf24166b2e, dataset head/last/64 samples, 96/96 hash vectors per pack,
and kernel.cu, program.metal, kernel.cl, program.h, memhard.h, memhard.metal byte-identical.
CPU verify 0.41 to 0.58 ms per warp (Swift 0.63 to 1.21), cache fill 175 to 181 ms one core.
CLI: bench, export, hash. program.json is written as valid JSON (the Swift quotes the
cache line mask inside the "item" string; fix pending in main.swift).

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-labs 2026-10-03 16:53:18 +00:00
parent 401f6af1df
commit ab99b67d3d
13 changed files with 2803 additions and 0 deletions

1
igneum-pow/.gitignore vendored Normal file
View file

@ -0,0 +1 @@
target/

105
igneum-pow/Cargo.lock generated Normal file
View file

@ -0,0 +1,105 @@
# This file is automatically @generated by Cargo.
# It is not intended for manual editing.
version = 4
[[package]]
name = "igneum-pow"
version = "0.1.0"
dependencies = [
"serde_json",
]
[[package]]
name = "itoa"
version = "1.0.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
[[package]]
name = "memchr"
version = "2.8.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
[[package]]
name = "proc-macro2"
version = "1.0.107"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
dependencies = [
"unicode-ident",
]
[[package]]
name = "quote"
version = "1.0.47"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
dependencies = [
"proc-macro2",
]
[[package]]
name = "serde"
version = "1.0.229"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
dependencies = [
"serde_core",
]
[[package]]
name = "serde_core"
version = "1.0.229"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
dependencies = [
"serde_derive",
]
[[package]]
name = "serde_derive"
version = "1.0.229"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
dependencies = [
"proc-macro2",
"quote",
"syn",
]
[[package]]
name = "serde_json"
version = "1.0.151"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
dependencies = [
"itoa",
"memchr",
"serde",
"serde_core",
"zmij",
]
[[package]]
name = "syn"
version = "3.0.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
dependencies = [
"proc-macro2",
"quote",
"unicode-ident",
]
[[package]]
name = "unicode-ident"
version = "1.0.26"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
[[package]]
name = "zmij"
version = "1.0.23"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"

30
igneum-pow/Cargo.toml Normal file
View file

@ -0,0 +1,30 @@
[package]
name = "igneum-pow"
version = "0.1.0"
edition = "2021"
description = "Igneum random-program GPU proof-of-work: seed, program generator, memory-hard dataset, CPU warp verifier and kernel emitters, bit-exact with proto-metal"
license = "MIT"
publish = false
[lib]
name = "igneum_pow"
path = "src/lib.rs"
[[bin]]
name = "igneum-pow"
path = "src/main.rs"
[dependencies]
[dev-dependencies]
serde_json = "1"
# The cache fill is 2^22 ChaCha12 blocks and the vector tests derive thousands of items.
# Unoptimised builds would make `cargo test` take minutes, so the dev profile is optimised too.
[profile.dev]
opt-level = 3
[profile.release]
opt-level = 3
lto = true
codegen-units = 1

97
igneum-pow/README.md Normal file
View file

@ -0,0 +1,97 @@
# igneum-pow
The Igneum lottery hash in Rust, bit-exact with the Swift prototype in `proto-metal/main.swift`. This is the
crate the rusty-kaspa fork will call (`docs/fork-map.md`, rows a1 to a3) so a node written in Rust can verify any
block and hand miners the kernel source for the epoch. No dependency outside the standard library; `serde_json`
is a dev-dependency for reading the packs in the tests.
Date: 3 October 2026. Toolchain: rustc 1.99.0 via rustup (the Homebrew 1.69 on PATH is too old; use
`~/.cargo/bin/cargo`).
## Modules
| Module | What it is | Swift namesake |
|---|---|---|
| `seed` | 32-byte seed words from a string (FNV-1a 64, four salts, finalised); `seed_words_from_bytes` is the boundary where the chain will feed the VDF output; SplitMix64 | `seedWords`, `SplitMix64` |
| `generator` | the 64-instruction program for a seed (op, dst, src, src2, imm, imm2, rot, bit, mask); levers `load_weight` and `wide_frac` | `generateProgram`, `GeneratorConfig` |
| `memhard` | 256 MiB cache (2^16 chains of 64 ChaCha12 blocks), mixer parameters, 8-round item derivation with the 32 lanes interleaved, `MemhardCpu::fetch` | `cpuFillCache`, `MixParams`, `deriveItems`, `MemhardCPU` |
| `verify` | the 32-lane warp interpreter, `DatasetMode::{ClosedForm, MemoryHard}`, `Epoch`, `hash_warp`, `verify_block` | `cpuWarp`, `DatasetSource` |
| `emit` | Metal, CUDA and OpenCL source, program.h, memhard.h, vectors.h, program.json, vectors.json, `export_pack` | `generateMSL`, `memhardMSL`, `emitMemhardCore`, `generateCUDA`, `generateOpenCL`, `exportPack` |
## The API the fork calls
```rust
use igneum_pow::{Epoch, DatasetMode};
// Once per epoch and day: generates the program and fills the 256 MiB cache (about 0.18 s on one core).
let epoch = Epoch::memory_hard("igneum-genesis", "2026-10-03");
let h: u64 = epoch.hash(nonce); // one nonce (computes its aligned 32-nonce warp)
let w: [u64; 32] = epoch.hash_warp(base_nonce); // one warp
let ok: bool = epoch.verify_block(nonce, target_u64);
// Miner programs for the epoch, byte-identical to the Swift exporter.
let pack = igneum_pow::emit::export_pack(&epoch, "2026-10-03", "igneum node");
pack.write_to(std::path::Path::new("out"))?; // kernel.cu, kernel.cl, program.metal, memhard.h, ...
```
`Epoch` is `Send + Sync`; build one and share it. `DatasetMode::ClosedForm` reproduces the two old packs
(`igneum-genesis`, `igneum-hourly`) and is not memory-hard. The hash is 64 bits; the fork maps it into its
256-bit target space in `consensus/pow/src/lib.rs`.
## CLI
```
cargo build --release
./target/release/igneum-pow bench --seed igneum-genesis [--warps 20] [--closed-form] [--day 2026-10-03]
./target/release/igneum-pow export --seed igneum-genesis --out <dir> [--closed-form]
./target/release/igneum-pow hash --seed igneum-genesis --nonce 4103
```
## Tests
`cargo test` (23 tests, 0.7 s after compile; the dev profile is optimised so the cache fill is quick):
| Check | Pack | Result |
|---|---|---|
| program.json instruction by instruction, op mix, loads per hash | igneum-genesis, igneum-genesis-mh, igneum-hourly | 3 x 64 match |
| Mixer parameters (key, rot, mul, rc) | igneum-genesis-mh | match |
| Cache head, last line, FNV-1a 64 `48c4f5bf24166b2e` | igneum-genesis-mh | match |
| Dataset head (16), `[MASK]`, 64 sampled words | all three | match |
| 96 hash vectors (3 warps x 32 lanes) | igneum-genesis-mh | 96/96 |
| 96 hash vectors | igneum-genesis, igneum-hourly | 96/96 each |
| kernel.cu, program.metal, kernel.cl, program.h byte-identical | all three | identical |
| memhard.h, memhard.metal byte-identical | igneum-genesis-mh | identical |
| program.json byte-identical (after the fix below) | all three | identical |
| vectors.json, vectors.h byte-identical apart from the provenance string | all three | identical |
An independent `diff -r` of `igneum-pow export` output against the checked-in packs shows the same two lines
only: the provenance string and the `"item"` line.
One deliberate difference: `proto-cuda/packs/igneum-genesis-mh/program.json` as written by the Swift is not
valid JSON (main.swift line 1291 uses `jhex` inside the `"item"` string, so the cache line mask is quoted inside a
quoted string). The Rust emitter writes `0x003fffff` bare; the test normalises that one line before comparing.
A node must hand miners valid JSON, so the Rust side does not reproduce the defect.
## Measured, 3 October 2026, Apple M5 Max, one core, release build
| Step | Rust | Swift (MEMHARD.md) |
|---|---|---|
| Cache fill, 256 MiB, 65,536 chains x 64 ChaCha12 blocks | 175 to 181 ms (5 quiet runs; 200 ms once with another build running) | 184.5 to 190.6 ms (C++ host reference 161.5) |
| CPU verify per warp, igneum-genesis, 104 loads, 3,328 items, avg of 20 | 0.441 ms | 0.649 ms |
| igneum-genesis/epoch1, 104 loads | 0.411 ms | 0.631 ms |
| igneum-genesis/epoch2, 112 loads | 0.488 ms | 0.701 ms |
| igneum-second-seed, 104 loads | 0.482 ms | 0.801 ms |
| igneum-second-seed/epoch1, 144 loads, 4,608 items | 0.579 ms | 1.205 ms |
| Cold single warps across the five seeds | 0.41 to 0.87 ms | 1.16 to 2.11 ms |
| Closed form, igneum-genesis | 0.002 ms | 0.017 ms |
The Rust verifier is 1.4x to 2.1x faster than the Swift one per warp; the registers are kept register-major
(`r[reg][lane]`) so the lane loops vectorise, and the item derivation interleaves the 32 lanes round by round as
the Swift does. The 10 ms gate holds with a margin of about 17x on the steady figure and 11x on the worst cold warp.
## Not done here
- No GPU. The vectors tie this crate to the Metal and CUDA results through the packs; nothing here runs a kernel.
- The epoch seed is still a string. `seed::seed_words_from_bytes` is where the VDF output will enter.
- The 256-bit target mapping and the `kaspa_pow::State` shape belong to the fork, not to this crate.

2
igneum-pow/rustfmt.toml Normal file
View file

@ -0,0 +1,2 @@
max_width = 120
use_small_heuristics = "Max"

1044
igneum-pow/src/emit.rs Normal file

File diff suppressed because it is too large Load diff

276
igneum-pow/src/generator.rs Normal file
View file

@ -0,0 +1,276 @@
//! The program generator: 64 integer instructions over 8 x u32 lane registers, run for 8 iterations.
//! Draw order, weights and the lever rules are those of `generateProgram` in `proto-metal/main.swift`.
use crate::seed::{program_rng, seed_words};
/// Iterations of the instruction list per hash.
pub const ITERATIONS: usize = 8;
/// Instructions per program.
pub const INSTR_COUNT: usize = 64;
/// Lanes per verification unit (one SIMD group / warp).
pub const LANES: usize = 32;
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
pub enum Op {
Add,
Sub,
Mul,
MulHi,
Xor,
Or,
Rotl,
Rotr,
Mad,
Shfl,
Load,
/// Warp-coalesced load (lever b). Never emitted unless `wide_frac > 0`.
WLoad,
}
impl Op {
/// The name used in program.json, kernel comments and the op mix.
pub fn name(self) -> &'static str {
match self {
Op::Add => "add",
Op::Sub => "sub",
Op::Mul => "mul",
Op::MulHi => "mulhi",
Op::Xor => "xor",
Op::Or => "or",
Op::Rotl => "rotl",
Op::Rotr => "rotr",
Op::Mad => "mad",
Op::Shfl => "shfl",
Op::Load => "load",
Op::WLoad => "wload",
}
}
pub fn from_name(s: &str) -> Option<Op> {
Some(match s {
"add" => Op::Add,
"sub" => Op::Sub,
"mul" => Op::Mul,
"mulhi" => Op::MulHi,
"xor" => Op::Xor,
"or" => Op::Or,
"rotl" => Op::Rotl,
"rotr" => Op::Rotr,
"mad" => Op::Mad,
"shfl" => Op::Shfl,
"load" => Op::Load,
"wload" => Op::WLoad,
_ => return None,
})
}
}
/// One instruction. Every field is drawn for every instruction whether the op uses it or not, so the
/// draw stream is identical for every op.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub struct Instr {
pub op: Op,
/// Destination register 0..7.
pub dst: u8,
/// Source register 0..7, never equal to `dst`.
pub src: u8,
/// Second source (mad only).
pub src2: u8,
/// Add immediate A.
pub imm: u32,
/// Add immediate B.
pub imm2: u32,
/// rotl amount 1..31.
pub rot: u32,
/// Selector bit of r0 for add, 0..31.
pub bit: u8,
/// Shuffle xor mask: 1, 2, 4, 8 or 16.
pub mask: u8,
}
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct Program {
pub seed_string: String,
pub seed: [u32; 8],
pub instrs: Vec<Instr>,
}
impl Program {
pub fn loads_per_hash(&self) -> usize {
self.instrs.iter().filter(|i| i.op == Op::Load || i.op == Op::WLoad).count() * ITERATIONS
}
pub fn wide_loads_per_hash(&self) -> usize {
self.instrs.iter().filter(|i| i.op == Op::WLoad).count() * ITERATIONS
}
pub fn has_wide(&self) -> bool {
self.instrs.iter().any(|i| i.op == Op::WLoad)
}
/// Distinct dataset items a 32-lane warp touches per hash: 32 per plain load, 2 per wide load.
pub fn items_per_warp(&self) -> usize {
(self.loads_per_hash() - self.wide_loads_per_hash()) * 32 + self.wide_loads_per_hash() * 2
}
/// Op histogram, count descending then name ascending, as the Swift prints it.
pub fn histogram(&self) -> Vec<(&'static str, usize)> {
let mut counts: Vec<(&'static str, usize)> = Vec::new();
for i in &self.instrs {
let name = i.op.name();
match counts.iter_mut().find(|(n, _)| *n == name) {
Some(e) => e.1 += 1,
None => counts.push((name, 1)),
}
}
counts.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(b.0)));
counts
}
/// "load=13 xor=13 ..." as written into program.h.
pub fn op_mix(&self) -> String {
self.histogram().iter().map(|(n, c)| format!("{n}={c}")).collect::<Vec<_>>().join(" ")
}
}
/// Weights sum to 100. Loads are 25 percent so the kernel leans on memory.
pub const OP_WEIGHTS: [(Op, u64); 11] = [
(Op::Load, 25),
(Op::Add, 12),
(Op::Xor, 10),
(Op::Mul, 8),
(Op::Mad, 8),
(Op::Shfl, 8),
(Op::Rotl, 7),
(Op::Sub, 6),
(Op::MulHi, 6),
(Op::Rotr, 6),
(Op::Or, 4),
];
/// Generator levers (MEMHARD.md section 2.4). The defaults reproduce the original generator exactly.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub struct GeneratorConfig {
/// Percent weight of the load op.
pub load_weight: u64,
/// Percent of load instructions emitted as warp-coalesced wide loads.
pub wide_frac: u64,
}
impl Default for GeneratorConfig {
fn default() -> Self {
Self { load_weight: 25, wide_frac: 0 }
}
}
impl GeneratorConfig {
/// Scaled weights: load gets `load_weight`, the other ten ops share the rest in their original
/// proportions, rounded by largest remainder so the table still sums to 100.
pub fn weights(&self) -> Vec<(Op, u64)> {
if self.load_weight == 25 {
return OP_WEIGHTS.to_vec();
}
let others = &OP_WEIGHTS[1..];
let total: u64 = others.iter().map(|w| w.1).sum(); // 75
let budget = 100 - self.load_weight;
let mut scaled: Vec<(Op, u64, u64)> =
others.iter().map(|&(op, w)| (op, (w * budget) / total, (w * budget) % total)).collect();
let mut sum: u64 = scaled.iter().map(|s| s.1).sum();
let mut order: Vec<usize> = (0..scaled.len()).collect();
order.sort_by(|&a, &b| scaled[b].2.cmp(&scaled[a].2).then_with(|| a.cmp(&b)));
let mut k = 0;
while sum < budget {
scaled[order[k]].1 += 1;
sum += 1;
k += 1;
}
let mut out = vec![(Op::Load, self.load_weight)];
out.extend(scaled.iter().map(|s| (s.0, s.1)));
out
}
}
/// The default generator for a seed string.
pub fn generate(seed_string: &str) -> Program {
generate_with(seed_string, &GeneratorConfig::default())
}
/// The generator with levers. `generateProgram` in the Swift, draw for draw.
pub fn generate_with(seed_string: &str, cfg: &GeneratorConfig) -> Program {
let seed = seed_words(seed_string);
generate_from_words(seed_string, seed, cfg)
}
/// The generator from already-derived seed words (what the chain will call once the VDF output is in).
pub fn generate_from_words(seed_string: &str, seed: [u32; 8], cfg: &GeneratorConfig) -> Program {
let mut rng = program_rng(&seed);
let weights = cfg.weights();
let mut instrs = Vec::with_capacity(INSTR_COUNT);
for _ in 0..INSTR_COUNT {
let mut roll = rng.below(100);
let mut op = Op::Add;
for &(o, w) in &weights {
if roll < w {
op = o;
break;
}
roll -= w;
}
let dst = rng.below(8);
let mut a = rng.below(7);
if a >= dst {
a += 1;
}
let b = rng.below(8);
let imm = rng.next() as u32;
let imm2 = rng.next() as u32;
let rot = 1 + rng.below(31) as u32;
let bit = rng.below(32);
let mask = 1u8 << rng.below(5);
// Lever (b): the already-drawn selector bit decides whether a load is wide, so the stream is unchanged.
if op == Op::Load && bit * 100 < cfg.wide_frac * 32 {
op = Op::WLoad;
}
instrs.push(Instr { op, dst: dst as u8, src: a as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask });
}
Program { seed_string: seed_string.to_string(), seed, instrs }
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn default_weights_unchanged() {
assert_eq!(GeneratorConfig::default().weights(), OP_WEIGHTS.to_vec());
}
#[test]
fn load_weight_17_table() {
// MEMHARD.md section 2.4: add=13 xor=11 mul=9 mad=9 shfl=9 rotl=8 sub=7 mulhi=7 rotr=6 or=4.
let w = GeneratorConfig { load_weight: 17, wide_frac: 0 }.weights();
let expect = [
(Op::Load, 17),
(Op::Add, 13),
(Op::Xor, 11),
(Op::Mul, 9),
(Op::Mad, 9),
(Op::Shfl, 9),
(Op::Rotl, 8),
(Op::Sub, 7),
(Op::MulHi, 7),
(Op::Rotr, 6),
(Op::Or, 4),
];
assert_eq!(w, expect.to_vec());
assert_eq!(w.iter().map(|x| x.1).sum::<u64>(), 100);
}
#[test]
fn genesis_shape() {
let p = generate("igneum-genesis");
assert_eq!(p.instrs.len(), 64);
assert_eq!(p.loads_per_hash(), 104);
assert_eq!(p.op_mix(), "load=13 xor=13 sub=7 shfl=6 add=5 mulhi=5 mad=4 rotr=4 mul=3 rotl=3 or=1");
for i in &p.instrs {
assert_ne!(i.dst, i.src);
assert!((1..=31).contains(&i.rot));
assert!(i.mask.is_power_of_two() && i.mask <= 16);
}
}
}

30
igneum-pow/src/lib.rs Normal file
View file

@ -0,0 +1,30 @@
//! igneum-pow: the Igneum lottery hash, bit-exact with the Swift prototype in `proto-metal/main.swift`.
//!
//! The crate has five parts, each mirroring one section of the prototype:
//!
//! * [`seed`]: the 32-byte seed words from a string (FNV-1a 64, four salts) and the SplitMix64 stream.
//! * [`generator`]: the 64-instruction program drawn from a seed.
//! * [`memhard`]: the 256 MiB ChaCha12 cache and the 8-round dataset item derivation (`proto-metal/MEMHARD.md`).
//! * [`verify`]: the 32-lane warp interpreter that computes the 64-bit hash on the CPU, deriving dataset
//! words on demand from the cache (or from the closed form, for the old packs).
//! * [`emit`]: the Metal, CUDA and OpenCL kernel text for a program, byte-identical to the Swift exporter.
//!
//! Nothing here depends on a crate outside the standard library. The integration points for the
//! rusty-kaspa fork (`docs/fork-map.md`) are [`verify::Epoch`], [`verify::Epoch::verify_block`] and
//! [`verify::Epoch::hash_warp`]; the node hands [`emit::Pack`] files to miners.
// The lane loops are written index style on purpose so they read like the kernels they mirror, and the
// SplitMix64 `next` keeps the Swift name.
#![allow(clippy::needless_range_loop, clippy::should_implement_trait, clippy::large_enum_variant)]
#![allow(clippy::manual_slice_size_calculation, clippy::too_many_arguments)]
pub mod emit;
pub mod generator;
pub mod memhard;
pub mod seed;
pub mod verify;
pub use generator::{generate, Instr, Op, Program};
pub use memhard::{Cache, MemhardCpu, MixParams};
pub use seed::{fnv1a64, seed_words, SplitMix64};
pub use verify::{hash_warp, verify_block, DatasetMode, DatasetSource, Epoch};

156
igneum-pow/src/main.rs Normal file
View file

@ -0,0 +1,156 @@
//! igneum-pow CLI.
//!
//! igneum-pow bench --seed <s> [--day <d>] [--closed-form] [--dataset-log2 28] [--warps 20]
//! igneum-pow export --seed <s> --out <dir> [--day <d>] [--closed-form] [--dataset-log2 28]
//! igneum-pow hash --seed <s> --nonce <n> [--day <d>] [--closed-form] [--dataset-log2 28]
use igneum_pow::emit::export_pack;
use igneum_pow::memhard::Cache;
use igneum_pow::seed::day_key;
use igneum_pow::verify::{DatasetMode, Epoch, DEFAULT_DATASET_LOG2};
use std::time::Instant;
struct Args {
cmd: String,
seed: String,
day: String,
out: Option<String>,
closed_form: bool,
dataset_log2: u32,
warps: usize,
nonce: u32,
}
fn usage() -> ! {
eprintln!(
"igneum-pow <bench|export|hash> --seed <string> [--day 2026-10-03] [--closed-form] [--dataset-log2 28]\n\
\x20 bench [--warps 20] fill the cache, then time the CPU verifier per 32-lane warp\n\
\x20 export --out <dir> write the program pack (kernel.cu, kernel.cl, program.metal, memhard.h, ...)\n\
\x20 hash --nonce <n> print the 64-bit hash of one nonce"
);
std::process::exit(2)
}
fn parse() -> Args {
let mut a = Args {
cmd: String::new(),
seed: "igneum-genesis".into(),
day: "2026-10-03".into(),
out: None,
closed_form: false,
dataset_log2: DEFAULT_DATASET_LOG2,
warps: 20,
nonce: 0,
};
let mut it = std::env::args().skip(1);
a.cmd = it.next().unwrap_or_else(|| usage());
while let Some(k) = it.next() {
let mut val = || it.next().unwrap_or_else(|| usage());
match k.as_str() {
"--seed" => a.seed = val(),
"--day" => a.day = val(),
"--out" => a.out = Some(val()),
"--closed-form" => a.closed_form = true,
"--dataset-log2" => a.dataset_log2 = val().parse().unwrap_or_else(|_| usage()),
"--warps" => a.warps = val().parse().unwrap_or_else(|_| usage()),
"--nonce" => a.nonce = val().parse().unwrap_or_else(|_| usage()),
_ => usage(),
}
}
a
}
fn main() {
let a = parse();
let mode = if a.closed_form { DatasetMode::ClosedForm } else { DatasetMode::MemoryHard };
match a.cmd.as_str() {
"bench" => bench(&a, mode),
"export" => export(&a, mode),
"hash" => {
let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2);
println!("{:016x}", e.hash(a.nonce));
}
_ => usage(),
}
}
fn bench(a: &Args, mode: DatasetMode) {
println!(
"igneum-pow bench: seed \"{}\", day \"{}\", dataset 2^{} words ({})",
a.seed,
a.day,
a.dataset_log2,
mode.name()
);
if mode == DatasetMode::MemoryHard {
// Time the cache fill on its own first (one core), then build the epoch (which fills it again).
let t0 = Instant::now();
let c = Cache::fill(day_key(&a.day));
let fill_ms = t0.elapsed().as_secs_f64() * 1e3;
println!("cache: fill {fill_ms:.1} ms on one core (2^26 words, 65536 chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", c.fnv1a64());
drop(c);
}
let t0 = Instant::now();
let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2);
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
println!(
"program: {} loads/hash, {} items/warp, op mix {}; epoch built in {build_ms:.1} ms",
e.program.loads_per_hash(),
e.program.items_per_warp(),
e.program.op_mix()
);
let bases = [0u32, 4096, 1_000_000];
for &b in &bases {
let t = Instant::now();
let r = e.interpret_warp(b);
let ms = t.elapsed().as_secs_f64() * 1e3;
println!(
"warp base {b}: single cold run {ms:.3} ms, {} items derived, lane0 {:016x} lane31 {:016x}",
r.items_derived, r.hashes[0], r.hashes[31]
);
}
let n = a.warps.max(1);
let t = Instant::now();
let mut sink = 0u64;
for i in 0..n {
let w = e.hash_warp((i as u32) * 32 + 65536);
sink ^= w[0];
}
let avg = t.elapsed().as_secs_f64() * 1e3 / n as f64;
println!("CPU verify: {avg:.3} ms per 32-lane warp, avg of {n} (checksum {sink:016x})");
}
fn export(a: &Args, mode: DatasetMode) {
let out = a.out.clone().unwrap_or_else(|| usage());
let t0 = Instant::now();
let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2);
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
println!("igneum-pow export {out}");
println!(
"seed \"{}\", day \"{}\", dataset 2^{} words ({}), loads/hash {}, wide loads/hash {}; epoch built in {build_ms:.1} ms",
a.seed,
a.day,
a.dataset_log2,
mode.name(),
e.program.loads_per_hash(),
e.program.wide_loads_per_hash()
);
println!("op mix: {}", e.program.op_mix());
let source = format!("igneum-pow (Rust) CPU interpreter, {} dataset", mode.name());
let pack = export_pack(&e, &a.day, &source);
let dir = std::path::Path::new(&out);
if let Err(err) = pack.write_to(dir) {
eprintln!("FAIL: write error {err}");
std::process::exit(1);
}
for (name, text) in &pack.files {
println!("wrote {}/{name} ({} bytes)", dir.display(), text.len());
}
for (i, b) in pack.bases.iter().enumerate() {
println!("vector warp base {b}: lane0 {:016x} lane31 {:016x}", pack.outs[i][0], pack.outs[i][31]);
}
if mode == DatasetMode::MemoryHard {
println!("cache FNV-1a 64 {:016x}", pack.vectors.cache_fnv);
}
println!("OVERALL: PASS (pack written)");
}

313
igneum-pow/src/memhard.rs Normal file
View file

@ -0,0 +1,313 @@
//! The memory-hard dataset of `proto-metal/MEMHARD.md`: a 256 MiB cache of chained ChaCha12 blocks keyed
//! by the day key, and 64-byte dataset items derived by 8 dependent cache reads through a seed-parameterised
//! ARX-multiply mixer. The verifier holds the cache and never the dataset.
//!
//! All arithmetic is on u32 modulo 2^32. Rotations are by 1..31 at every call site.
use crate::seed::{day_key, fnv1a64_words, SplitMix64};
pub const CACHE_LOG2_WORDS: usize = 26;
pub const CACHE_SEGMENT_LOG2_LINES: usize = 6;
/// 2^26 words = 256 MiB.
pub const CACHE_WORDS: usize = 1 << CACHE_LOG2_WORDS;
/// 2^22 lines of 16 words.
pub const CACHE_LINES: usize = CACHE_WORDS >> 4;
/// 64 chained lines per segment.
pub const CACHE_LINES_PER_SEGMENT: usize = 1 << CACHE_SEGMENT_LOG2_LINES;
/// 2^16 independent segments.
pub const CACHE_SEGMENTS: usize = CACHE_LINES >> CACHE_SEGMENT_LOG2_LINES;
pub const CACHE_LINE_MASK: u32 = (CACHE_LINES - 1) as u32;
pub const ITEM_ROUNDS: usize = 8;
pub const CHACHA_ROUNDS: usize = 12;
/// The ChaCha constants "expand 32-byte k".
pub const CHACHA_SIGMA: [u32; 4] = [0x61707865, 0x3320646e, 0x79622d32, 0x6b206574];
/// "Igne", "umMH".
pub const CACHE_TAG: [u32; 2] = [0x49676e65, 0x756d4d48];
#[inline(always)]
fn rotl(x: u32, n: u32) -> u32 {
x.rotate_left(n)
}
/// The ChaCha quarter round with explicit rotations.
#[inline(always)]
fn qr(s: &mut [u32; 16], a: usize, b: usize, c: usize, d: usize, r1: u32, r2: u32, r3: u32, r4: u32) {
s[a] = s[a].wrapping_add(s[b]);
s[d] ^= s[a];
s[d] = rotl(s[d], r1);
s[c] = s[c].wrapping_add(s[d]);
s[b] ^= s[c];
s[b] = rotl(s[b], r2);
s[a] = s[a].wrapping_add(s[b]);
s[d] ^= s[a];
s[d] = rotl(s[d], r3);
s[c] = s[c].wrapping_add(s[d]);
s[b] ^= s[c];
s[b] = rotl(s[b], r4);
}
/// `y = ChaCha12 core(x) + x`. Standard rotations 16, 12, 8, 7; column round then diagonal round, six times.
#[inline]
pub fn chacha_block(x: &[u32; 16]) -> [u32; 16] {
let mut y = *x;
for _ in 0..CHACHA_ROUNDS / 2 {
qr(&mut y, 0, 4, 8, 12, 16, 12, 8, 7);
qr(&mut y, 1, 5, 9, 13, 16, 12, 8, 7);
qr(&mut y, 2, 6, 10, 14, 16, 12, 8, 7);
qr(&mut y, 3, 7, 11, 15, 16, 12, 8, 7);
qr(&mut y, 0, 5, 10, 15, 16, 12, 8, 7);
qr(&mut y, 1, 6, 11, 12, 16, 12, 8, 7);
qr(&mut y, 2, 7, 8, 13, 16, 12, 8, 7);
qr(&mut y, 3, 4, 9, 14, 16, 12, 8, 7);
}
for i in 0..16 {
y[i] = y[i].wrapping_add(x[i]);
}
y
}
/// Mixer parameters drawn from the day key. Draw order: ROT[0..7] (1..31), MUL[0..15] (odd), RC[0..15].
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct MixParams {
pub key: [u32; 8],
pub rot: [u32; 8],
pub mul: [u32; 16],
pub rc: [u32; 16],
}
impl MixParams {
pub fn new(key: [u32; 8]) -> Self {
let mut rng = SplitMix64::new(key[0] as u64 | ((key[1] as u64) << 32));
let mut rot = [0u32; 8];
let mut mul = [0u32; 16];
let mut rc = [0u32; 16];
for r in rot.iter_mut() {
*r = 1 + rng.below(31) as u32;
}
for m in mul.iter_mut() {
*m = (rng.next() as u32) | 1;
}
for c in rc.iter_mut() {
*c = rng.next() as u32;
}
Self { key, rot, mul, rc }
}
/// Parameters for a day string: the key is `seed_words("day/" + day)`.
pub fn for_day(day: &str) -> Self {
Self::new(day_key(day))
}
}
/// Round key `(r + 1) * 0x9E3779B9` mod 2^32.
#[inline(always)]
pub fn round_key(r: usize) -> u32 {
((r + 1) as u32).wrapping_mul(0x9E3779B9)
}
/// `M_r` on 16 words in place: per word `(s ^ (RC + rk)) * MUL`, then one ChaCha-shaped double round with
/// the four column rotations `ROT[0..3]` and the four diagonal rotations `ROT[4..7]`.
#[inline(always)]
pub fn mixer(s: &mut [u32; 16], rk: u32, mp: &MixParams) {
for i in 0..16 {
s[i] = (s[i] ^ mp.rc[i].wrapping_add(rk)).wrapping_mul(mp.mul[i]);
}
let r = &mp.rot;
qr(s, 0, 4, 8, 12, r[0], r[1], r[2], r[3]);
qr(s, 1, 5, 9, 13, r[0], r[1], r[2], r[3]);
qr(s, 2, 6, 10, 14, r[0], r[1], r[2], r[3]);
qr(s, 3, 7, 11, 15, r[0], r[1], r[2], r[3]);
qr(s, 0, 5, 10, 15, r[4], r[5], r[6], r[7]);
qr(s, 1, 6, 11, 12, r[4], r[5], r[6], r[7]);
qr(s, 2, 7, 8, 13, r[4], r[5], r[6], r[7]);
qr(s, 3, 4, 9, 14, r[4], r[5], r[6], r[7]);
}
/// The 256 MiB cache for one day key.
pub struct Cache {
pub key: [u32; 8],
words: Vec<u32>,
}
impl Cache {
/// One segment: 64 chained lines written at `cache[seg * 1024 ..]`.
/// `in_j = prev XOR (sigma || K || seg || j || tag)`, `line_j = B(in_j)`, `prev_0 = 0`.
pub fn fill_segment(words: &mut [u32], seg: usize, key: &[u32; 8]) {
let base = (seg << CACHE_SEGMENT_LOG2_LINES) * 16;
let seg_words = &mut words[base..base + CACHE_LINES_PER_SEGMENT * 16];
let mut prev = [0u32; 16];
for (j, line) in seg_words.as_chunks_mut::<16>().0.iter_mut().enumerate() {
let mut x = [0u32; 16];
x[..4].copy_from_slice(&CHACHA_SIGMA);
x[4..12].copy_from_slice(key);
x[12] = seg as u32;
x[13] = j as u32;
x[14] = CACHE_TAG[0];
x[15] = CACHE_TAG[1];
for i in 0..16 {
x[i] ^= prev[i];
}
let y = chacha_block(&x);
line.copy_from_slice(&y);
prev = y;
}
}
/// The whole cache on the calling thread: 65,536 chains of 64 ChaCha12 blocks, in segment order.
pub fn fill(key: [u32; 8]) -> Cache {
let mut words = vec![0u32; CACHE_WORDS];
for seg in 0..CACHE_SEGMENTS {
Self::fill_segment(&mut words, seg, &key);
}
Cache { key, words }
}
pub fn for_day(day: &str) -> Cache {
Self::fill(day_key(day))
}
#[inline(always)]
pub fn words(&self) -> &[u32] {
&self.words
}
/// Cache line `a` (0 <= a < 2^22) as 16 words.
#[inline(always)]
pub fn line(&self, a: u32) -> &[u32] {
let o = (a & CACHE_LINE_MASK) as usize * 16;
&self.words[o..o + 16]
}
/// FNV-1a 64 over the cache as little-endian bytes (what `vectors.h` carries as `IGNEUM_CACHE_FNV64`).
pub fn fnv1a64(&self) -> u64 {
fnv1a64_words(&self.words)
}
}
/// Derive `ts.len()` items into `out`, all chains interleaved round by round so the cache-line misses of
/// independent items overlap in the memory system (`deriveItems` in the Swift).
pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) {
let n = ts.len();
debug_assert!(out.len() >= n);
for k in 0..n {
let s = &mut out[k];
let t = ts[k];
s[..8].copy_from_slice(&mp.key);
for i in 0..8 {
s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]);
}
}
for r in 0..ITEM_ROUNDS {
let rk = round_key(r);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
}
for s in out[..n].iter_mut() {
let line = cache.line(s[0]);
for i in 0..16 {
s[i] ^= line[i];
}
}
}
let rk = round_key(ITEM_ROUNDS);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
}
}
/// One dataset item, 16 words.
pub fn derive_item(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] {
let mut out = [[0u32; 16]; 1];
derive_items(&[t], mp, cache, &mut out);
out[0]
}
/// The CPU verifier's view of the memory-hard dataset: the mixer parameters and the 256 MiB cache.
pub struct MemhardCpu {
pub params: MixParams,
pub cache: Cache,
}
/// Largest batch `MemhardCpu::fetch` accepts (two warps).
pub const FETCH_MAX: usize = 64;
impl MemhardCpu {
pub fn new(key: [u32; 8]) -> Self {
Self { params: MixParams::new(key), cache: Cache::fill(key) }
}
pub fn for_day(day: &str) -> Self {
Self::new(day_key(day))
}
/// `dataset[w] = item(w >> 4)[w & 15]`.
pub fn word(&self, w: u32) -> u32 {
derive_item(w >> 4, &self.params, &self.cache)[(w & 15) as usize]
}
/// `out[k] = dataset[idx[k]]` for every k, `idx.len() <= FETCH_MAX`. Equal items are derived once.
/// Returns the number of distinct items derived.
pub fn fetch(&self, idx: &[u32], out: &mut [u32]) -> usize {
let n = idx.len();
assert!(n <= FETCH_MAX && out.len() >= n);
let mut uniq = [0u32; FETCH_MAX];
let mut slot = [0u8; FETCH_MAX];
let mut u = 0usize;
for k in 0..n {
let t = idx[k] >> 4;
let found = uniq[..u].iter().position(|&x| x == t);
let j = match found {
Some(j) => j,
None => {
uniq[u] = t;
u += 1;
u - 1
}
};
slot[k] = j as u8;
}
let mut items = [[0u32; 16]; FETCH_MAX];
derive_items(&uniq[..u], &self.params, &self.cache, &mut items);
for k in 0..n {
out[k] = items[slot[k] as usize][(idx[k] & 15) as usize];
}
u
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn mix_params_for_day() {
// MEMHARD.md section 1.4 and the igneum-genesis-mh pack.
let mp = MixParams::for_day("2026-10-03");
assert_eq!(mp.rot, [20, 20, 19, 4, 26, 3, 3, 27]);
assert_eq!(mp.mul[0], 0x42146205);
assert_eq!(mp.mul[15], 0x99cfb423);
assert_eq!(mp.rc[0], 0xbab68293);
assert_eq!(mp.rc[15], 0x31b49ee2);
assert!(mp.mul.iter().all(|m| m & 1 == 1));
}
#[test]
fn chacha_block_is_a_permutation_plus_feedforward() {
let x = [1u32; 16];
let y = chacha_block(&x);
assert_ne!(x, y);
let z = chacha_block(&x);
assert_eq!(y, z);
}
#[test]
fn first_cache_line_matches_pack() {
// vectors.json cache_head for day 2026-10-03: segment 0, line 0, with prev = 0.
let key = day_key("2026-10-03");
let mut words = vec![0u32; CACHE_LINES_PER_SEGMENT * 16];
Cache::fill_segment(&mut words, 0, &key);
assert_eq!(
&words[..16],
&[
0x355a86d2, 0x7957db1c, 0xd21772af, 0x6fc1e09b, 0xd55ce61d, 0x6e6a278b, 0xd3f543ce, 0x223d8e82,
0x143ab337, 0x2e9f05bd, 0x2eb389bf, 0x0c6e449e, 0x5cfa4222, 0xba6560fe, 0x8e3e1aa4, 0xdbcc1d53
]
);
}
}

114
igneum-pow/src/seed.rs Normal file
View file

@ -0,0 +1,114 @@
//! Seed derivation and the SplitMix64 stream, exactly as `proto-metal/main.swift` does them.
//!
//! Today a seed is a string ("igneum-genesis", "day/2026-10-03"). On the chain the epoch seed will be the
//! output of a class-group VDF over a certified checkpoint hash. [`seed_words_from_bytes`] is the function
//! boundary for that: whatever bytes the chain settles on go through the same FNV-1a construction, so the
//! generator and the day key never need to know where the bytes came from.
/// FNV-1a 64 over `bytes` with the standard basis. Used for the cache fingerprint in the packs.
pub fn fnv1a64(bytes: &[u8]) -> u64 {
fnv1a64_with_basis(0xcbf29ce484222325, bytes)
}
#[inline]
fn fnv1a64_with_basis(basis: u64, bytes: &[u8]) -> u64 {
let mut h = basis;
for &b in bytes {
h ^= b as u64;
h = h.wrapping_mul(0x100000001b3);
}
h
}
/// FNV-1a 64 over 32-bit words in little-endian byte order (the cache is hashed as raw memory).
pub fn fnv1a64_words(words: &[u32]) -> u64 {
let mut h: u64 = 0xcbf29ce484222325;
for &w in words {
for b in w.to_le_bytes() {
h ^= b as u64;
h = h.wrapping_mul(0x100000001b3);
}
}
h
}
/// The 32-byte seed (8 x u32) from arbitrary bytes: FNV-1a 64 with four salts, each finalised with the
/// murmur-style mix `h ^= h >> 33; h *= 0xff51afd7ed558ccd; h ^= h >> 33`; low word then high word.
pub fn seed_words_from_bytes(bytes: &[u8]) -> [u32; 8] {
let mut words = [0u32; 8];
for salt in 0..4u64 {
let basis = 0xcbf29ce484222325u64 ^ salt.wrapping_mul(0x9E3779B97F4A7C15);
let mut h = fnv1a64_with_basis(basis, bytes);
h ^= h >> 33;
h = h.wrapping_mul(0xff51afd7ed558ccd);
h ^= h >> 33;
words[2 * salt as usize] = h as u32;
words[2 * salt as usize + 1] = (h >> 32) as u32;
}
words
}
/// The 32-byte seed from a string (its UTF-8 bytes). `seedWords` in the Swift.
pub fn seed_words(s: &str) -> [u32; 8] {
seed_words_from_bytes(s.as_bytes())
}
/// The day key: the 8 words of `seed_words("day/" + day)`. `K` in MEMHARD.md; `d0, d1` are `K[0], K[1]`.
pub fn day_key(day: &str) -> [u32; 8] {
seed_words(&format!("day/{day}"))
}
/// SplitMix64, the one deterministic stream every draw in the prototype comes from.
#[derive(Clone, Copy, Debug)]
pub struct SplitMix64 {
pub s: u64,
}
impl SplitMix64 {
pub fn new(s: u64) -> Self {
Self { s }
}
#[inline]
pub fn next(&mut self) -> u64 {
self.s = self.s.wrapping_add(0x9E3779B97F4A7C15);
let mut z = self.s;
z = (z ^ (z >> 30)).wrapping_mul(0xBF58476D1CE4E5B9);
z = (z ^ (z >> 27)).wrapping_mul(0x94D049BB133111EB);
z ^ (z >> 31)
}
/// `next() % n` as the Swift `below` does it (modulo, not rejection sampling).
#[inline]
pub fn below(&mut self, n: u64) -> u64 {
self.next() % n
}
}
/// The generator's stream for a seed: `(w0 | w1 << 32) ^ ((w2 | w3 << 32) * 0x9E3779B97F4A7C15)`.
pub fn program_rng(seed: &[u32; 8]) -> SplitMix64 {
let lo = seed[0] as u64 | ((seed[1] as u64) << 32);
let hi = seed[2] as u64 | ((seed[3] as u64) << 32);
SplitMix64::new(lo ^ hi.wrapping_mul(0x9E3779B97F4A7C15))
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn genesis_seed_words() {
// From proto-cuda/packs/igneum-genesis/program.json.
assert_eq!(
seed_words("igneum-genesis"),
[0x67a9a7be, 0x1a155b25, 0xfddfb732, 0x4b5af2e8, 0xc55caf33, 0xa27c13b7, 0x06628a48, 0x03852469]
);
}
#[test]
fn day_key_2026_10_03() {
// MEMHARD.md section 1.1.
assert_eq!(
day_key("2026-10-03"),
[0x3067619f, 0x3c269176, 0x84a03b03, 0xf8c63294, 0xff977c5b, 0xe60def3e, 0x63630141, 0xb8fbcb58]
);
}
}

349
igneum-pow/src/verify.rs Normal file
View file

@ -0,0 +1,349 @@
//! The CPU reference interpreter for one 32-lane warp (`cpuWarpTraced` in the Swift) and the API the node
//! calls. Dataset words come from the memory-hard cache (default) or from the closed form (old packs).
use crate::generator::{generate, Instr, Op, Program, ITERATIONS, LANES};
use crate::memhard::MemhardCpu;
use crate::seed::day_key;
/// Dataset element, closed form of (day words, index). The original prototype's six-operation element.
#[inline(always)]
pub fn dataset_elem(i: u32, d0: u32, d1: u32) -> u32 {
let mut x = i ^ d0;
x = x.wrapping_mul(0x9E3779B1);
x ^= x >> 15;
x = x.wrapping_add(d1);
x = x.wrapping_mul(0x85EBCA77);
x ^= x >> 13;
x = x.wrapping_mul(0xC2B2AE3D);
x ^= x >> 16;
x
}
/// splitmix32, used for the register init.
#[inline(always)]
pub fn splitmix32(v: u32) -> u32 {
let mut x = v;
x ^= x >> 16;
x = x.wrapping_mul(0x7feb352d);
x ^= x >> 15;
x = x.wrapping_mul(0x846ca68b);
x ^= x >> 16;
x
}
/// Which construction fills the dataset words.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum DatasetMode {
/// The six-operation closed form (packs igneum-genesis and igneum-hourly). Not memory-hard.
ClosedForm,
/// The 256 MiB cache and 8 dependent reads per item (pack igneum-genesis-mh, MEMHARD.md). The default.
MemoryHard,
}
impl DatasetMode {
pub fn name(self) -> &'static str {
match self {
DatasetMode::ClosedForm => "closed-form",
DatasetMode::MemoryHard => "memory-hard",
}
}
}
/// Where the interpreter reads dataset words from.
pub enum Dataset {
ClosedForm { d0: u32, d1: u32 },
MemoryHard(MemhardCpu),
}
/// A dataset of `2^log2` words plus the construction that fills it.
pub struct DatasetSource {
pub log2_words: u32,
pub mask: u32,
/// The day key `K`; `d0, d1 = K[0], K[1]`.
pub key: [u32; 8],
pub dataset: Dataset,
}
impl DatasetSource {
/// Build the source for a day. Memory-hard mode fills the 256 MiB cache on the calling thread.
pub fn new(day: &str, mode: DatasetMode, log2_words: u32) -> Self {
Self::from_key(day_key(day), mode, log2_words)
}
pub fn from_key(key: [u32; 8], mode: DatasetMode, log2_words: u32) -> Self {
assert!((4..=32).contains(&log2_words), "dataset log2 must be in 4..=32");
let mask = if log2_words == 32 { u32::MAX } else { (1u32 << log2_words) - 1 };
let dataset = match mode {
DatasetMode::ClosedForm => Dataset::ClosedForm { d0: key[0], d1: key[1] },
DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::new(key)),
};
Self { log2_words, mask, key, dataset }
}
pub fn mode(&self) -> DatasetMode {
match self.dataset {
Dataset::ClosedForm { .. } => DatasetMode::ClosedForm,
Dataset::MemoryHard(_) => DatasetMode::MemoryHard,
}
}
pub fn memhard(&self) -> Option<&MemhardCpu> {
match &self.dataset {
Dataset::MemoryHard(m) => Some(m),
_ => None,
}
}
/// `dataset[w & mask]`.
pub fn word(&self, w: u32) -> u32 {
let w = w & self.mask;
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => dataset_elem(w, *d0, *d1),
Dataset::MemoryHard(m) => m.word(w),
}
}
/// `out[k] = dataset[idx[k]]`; indices are already masked. Returns items derived (0 for the closed form).
#[inline]
fn fetch(&self, idx: &[u32; LANES], out: &mut [u32; LANES]) -> usize {
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => {
for k in 0..LANES {
out[k] = dataset_elem(idx[k], *d0, *d1);
}
0
}
Dataset::MemoryHard(m) => m.fetch(idx, out),
}
}
}
/// The result of interpreting one warp.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub struct WarpResult {
pub hashes: [u64; LANES],
/// Distinct dataset items derived from the cache (0 in closed-form mode).
pub items_derived: usize,
}
#[inline(always)]
fn mulhi32(a: u32, b: u32) -> u32 {
((a as u64 * b as u64) >> 32) as u32
}
/// Interpret `program` for the 32 nonces `base_nonce .. base_nonce + 31` (wrapping). Registers are kept
/// register-major (`r[reg][lane]`) so the per-lane loops vectorise; the semantics are those of the Metal
/// and CUDA kernels instruction for instruction.
pub fn interpret_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) -> WarpResult {
let mask = ds.mask;
let seed = &program.seed;
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
let nonce = base_nonce.wrapping_add(lane as u32);
for i in 0..8 {
let mut x = nonce ^ seed[i];
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
x = splitmix32(x);
r[i][lane] = x ^ seed[(i + 1) & 7];
}
}
let mut items_derived = 0usize;
let mut idx = [0u32; LANES];
let mut val = [0u32; LANES];
for _ in 0..ITERATIONS {
let sel = r[0];
for ins in &program.instrs {
step(ins, &mut r, &sel, mask, ds, &mut idx, &mut val, &mut items_derived);
}
}
let mut hashes = [0u64; LANES];
for lane in 0..LANES {
let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21);
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
hashes[lane] = ((hi as u64) << 32) | lo as u64;
}
WarpResult { hashes, items_derived }
}
#[inline(always)]
#[allow(clippy::too_many_arguments)]
fn step(
ins: &Instr,
r: &mut [[u32; LANES]; 8],
sel: &[u32; LANES],
mask: u32,
ds: &DatasetSource,
idx: &mut [u32; LANES],
val: &mut [u32; LANES],
items_derived: &mut usize,
) {
let d = ins.dst as usize;
let a = ins.src as usize;
match ins.op {
Op::Add => {
let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32);
let src = r[a];
for lane in 0..LANES {
let s = (sel[lane] >> bit) & 1;
let c = if s != 0 { imm2 } else { imm };
r[d][lane] = r[d][lane].wrapping_add(src[lane]).wrapping_add(c);
}
}
Op::Sub => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_sub(src[lane]);
}
}
Op::Mul => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_mul(src[lane]);
}
}
Op::MulHi => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = mulhi32(r[d][lane], src[lane]);
}
}
Op::Xor => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] ^= src[lane];
}
}
Op::Or => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] |= src[lane];
}
}
Op::Rotl => {
let n = ins.rot;
for lane in 0..LANES {
r[d][lane] = r[d][lane].rotate_left(n);
}
}
Op::Rotr => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].rotate_right(src[lane] & 31);
}
}
Op::Mad => {
let src = r[a];
let src2 = r[ins.src2 as usize];
for lane in 0..LANES {
r[d][lane] = src[lane].wrapping_mul(src2[lane]).wrapping_add(r[d][lane]);
}
}
Op::Shfl => {
let src = r[a];
let m = ins.mask as usize;
for lane in 0..LANES {
r[d][lane] ^= src[lane ^ m];
}
}
Op::Load => {
for lane in 0..LANES {
idx[lane] = r[a][lane] & mask;
}
*items_derived += ds.fetch(idx, val);
for lane in 0..LANES {
r[d][lane] ^= val[lane];
}
}
Op::WLoad => {
// Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l.
let base = (r[a][0] & mask) & !31;
for lane in 0..LANES {
idx[lane] = base + lane as u32;
}
*items_derived += ds.fetch(idx, val);
for lane in 0..LANES {
r[d][lane] ^= val[lane];
}
}
}
}
/// The 32 hashes of one warp (`cpuWarp` in the Swift).
pub fn hash_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) -> [u64; LANES] {
interpret_warp(program, base_nonce, ds).hashes
}
/// Everything a node needs to verify blocks of one epoch on one day: the program for the epoch seed and the
/// dataset source for the day key. Building one in memory-hard mode fills the 256 MiB cache (about 0.2 s on
/// one core); keep it for the whole epoch and share it between threads (`&Epoch` is `Send + Sync`).
pub struct Epoch {
pub program: Program,
pub dataset: DatasetSource,
}
/// Default dataset size: 2^28 words = 1 GiB.
pub const DEFAULT_DATASET_LOG2: u32 = 28;
impl Epoch {
pub fn new(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32) -> Self {
Self { program: generate(seed), dataset: DatasetSource::new(day, mode, dataset_log2) }
}
/// The production shape: memory-hard, 1 GiB dataset.
pub fn memory_hard(seed: &str, day: &str) -> Self {
Self::new(seed, day, DatasetMode::MemoryHard, DEFAULT_DATASET_LOG2)
}
/// The 32 hashes of the warp starting at `base_nonce`.
pub fn hash_warp(&self, base_nonce: u32) -> [u64; LANES] {
hash_warp(&self.program, base_nonce, &self.dataset)
}
pub fn interpret_warp(&self, base_nonce: u32) -> WarpResult {
interpret_warp(&self.program, base_nonce, &self.dataset)
}
/// The hash of one nonce. The verification unit is a warp, so the 31 sibling nonces of the aligned
/// 32-nonce group are computed too (the shuffles couple the lanes).
pub fn hash(&self, nonce: u32) -> u64 {
self.hash_warp(nonce & !31)[(nonce & 31) as usize]
}
/// `hash(nonce) <= target`. The hash is 64 bits; the fork maps it into its 256-bit target space.
pub fn verify_block(&self, nonce: u32, target: u64) -> bool {
self.hash(nonce) <= target
}
}
/// One-shot `verify_block(seed, nonce, target)`: builds the epoch (cache fill included) and checks. For a
/// node use [`Epoch`] and keep it; this exists for scripts and tests.
pub fn verify_block(seed: &str, day: &str, nonce: u32, target: u64) -> bool {
Epoch::memory_hard(seed, day).verify_block(nonce, target)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn closed_form_head_matches_pack() {
// vectors.json dataset_head for igneum-genesis (closed form), day 2026-10-03.
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28);
assert_eq!(ds.word(0), 0x82174c0f);
assert_eq!(ds.word(1), 0x577bdb9c);
assert_eq!(ds.word(0x0fffffff), 0xf78c84a4);
}
#[test]
fn closed_form_genesis_vector_lane0() {
let e = Epoch::new("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 28);
let w = e.hash_warp(0);
assert_eq!(w[0], 0x2941e93c76cb1910);
assert_eq!(w[31], 0x453388e1be04e25f);
assert_eq!(e.hash(0), w[0]);
assert_eq!(e.hash(31), w[31]);
assert!(e.verify_block(0, u64::MAX));
assert!(e.verify_block(0, w[0]));
assert!(!e.verify_block(0, w[0] - 1));
}
}

286
igneum-pow/tests/packs.rs Normal file
View file

@ -0,0 +1,286 @@
//! Agreement with the Swift prototype through the checked-in packs under proto-cuda/packs/.
//!
//! igneum-genesis-mh: memory-hard dataset (cache FNV, head and last line, 64 sampled words, 96 hashes).
//! igneum-genesis and igneum-hourly: closed-form dataset (head, last, 64 samples, 96 hashes each).
//! All three: program.json instruction by instruction, and every emitted source file byte for byte.
use igneum_pow::emit::{
cuda_kernel, cuda_memhard_header, export_pack, metal_memhard, metal_program, opencl_kernel, program_header,
program_json, LoadSource,
};
use igneum_pow::generator::{generate, Op};
use igneum_pow::memhard::CACHE_WORDS;
use igneum_pow::verify::{DatasetMode, Epoch};
use serde_json::Value;
use std::path::PathBuf;
use std::sync::OnceLock;
const DAY: &str = "2026-10-03";
fn packs_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs")
}
fn read(pack: &str, file: &str) -> String {
let p = packs_dir().join(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
/// The Swift exporter writes the cache line mask quoted inside the "item" string, which is not valid JSON.
/// Normalise that one defect so the file can be parsed and compared; the Rust emitter writes it bare.
fn fix_swift_item_line(s: &str) -> String {
s.replace("& \"0x003fffff\";", "& 0x003fffff;")
}
fn json(pack: &str, file: &str) -> Value {
let text = fix_swift_item_line(&read(pack, file));
serde_json::from_str(&text).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"))
}
fn hex32(v: &Value) -> u32 {
u32::from_str_radix(v.as_str().unwrap().trim_start_matches("0x"), 16).unwrap()
}
fn hex64(v: &Value) -> u64 {
u64::from_str_radix(v.as_str().unwrap().trim_start_matches("0x"), 16).unwrap()
}
/// One memory-hard epoch shared by every test (the cache fill is 256 MiB and about 0.2 s).
fn mh_epoch() -> &'static Epoch {
static E: OnceLock<Epoch> = OnceLock::new();
E.get_or_init(|| Epoch::new("igneum-genesis", DAY, DatasetMode::MemoryHard, 28))
}
fn closed_epoch(seed: &str) -> Epoch {
Epoch::new(seed, DAY, DatasetMode::ClosedForm, 28)
}
fn check_program_json(pack: &str) {
let j = json(pack, "program.json");
let seed = j["seed"].as_str().unwrap();
let p = generate(seed);
let sw: Vec<u32> = j["seed_words"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(p.seed.to_vec(), sw, "{pack}: seed words");
assert_eq!(p.loads_per_hash() as u64, j["loads_per_hash"].as_u64().unwrap(), "{pack}: loads per hash");
let instrs = j["instructions"].as_array().unwrap();
assert_eq!(instrs.len(), p.instrs.len(), "{pack}: instruction count");
for (k, (ins, ji)) in p.instrs.iter().zip(instrs).enumerate() {
assert_eq!(ji["i"].as_u64().unwrap() as usize, k);
assert_eq!(Op::from_name(ji["op"].as_str().unwrap()).unwrap(), ins.op, "{pack} #{k} op");
assert_eq!(ji["dst"].as_u64().unwrap(), ins.dst as u64, "{pack} #{k} dst");
assert_eq!(ji["src"].as_u64().unwrap(), ins.src as u64, "{pack} #{k} src");
assert_eq!(ji["src2"].as_u64().unwrap(), ins.src2 as u64, "{pack} #{k} src2");
assert_eq!(hex32(&ji["imm"]), ins.imm, "{pack} #{k} imm");
assert_eq!(hex32(&ji["imm2"]), ins.imm2, "{pack} #{k} imm2");
assert_eq!(ji["rot"].as_u64().unwrap(), ins.rot as u64, "{pack} #{k} rot");
assert_eq!(ji["bit"].as_u64().unwrap(), ins.bit as u64, "{pack} #{k} bit");
assert_eq!(ji["mask"].as_u64().unwrap(), ins.mask as u64, "{pack} #{k} mask");
}
// Op mix.
let mix = j["op_mix"].as_object().unwrap();
for (name, count) in p.histogram() {
assert_eq!(mix[name].as_u64().unwrap() as usize, count, "{pack}: op_mix {name}");
}
assert_eq!(mix.len(), p.histogram().len());
}
#[test]
fn program_json_matches_all_packs() {
for pack in ["igneum-genesis", "igneum-genesis-mh", "igneum-hourly"] {
check_program_json(pack);
}
}
#[test]
fn mixer_params_match_pack() {
let j = json("igneum-genesis-mh", "program.json");
let mp = &mh_epoch().dataset.memhard().unwrap().params;
let key: Vec<u32> = j["dataset"]["key"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(mp.key.to_vec(), key);
let rot: Vec<u32> =
j["dataset"]["mixer"]["rot"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u32).collect();
assert_eq!(mp.rot.to_vec(), rot);
let mul: Vec<u32> = j["dataset"]["mixer"]["mul"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(mp.mul.to_vec(), mul);
let rc: Vec<u32> = j["dataset"]["mixer"]["rc"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(mp.rc.to_vec(), rc);
assert_eq!(hex32(&j["dataset"]["d0"]), mp.key[0]);
assert_eq!(hex32(&j["dataset"]["d1"]), mp.key[1]);
}
#[test]
fn cache_matches_vectors() {
let v = json("igneum-genesis-mh", "vectors.json");
let m = mh_epoch().dataset.memhard().unwrap();
let w = m.cache.words();
assert_eq!(w.len(), CACHE_WORDS);
let head: Vec<u32> = v["cache_head"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(&w[..16], &head[..]);
let last: Vec<u32> = v["cache_last_line"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(&w[CACHE_WORDS - 16..], &last[..]);
assert_eq!(m.cache.fnv1a64(), 0x48c4f5bf24166b2e, "cache FNV-1a 64 (MEMHARD.md)");
assert_eq!(m.cache.fnv1a64(), hex64(&v["cache_fnv1a64"]));
}
fn check_dataset_words(pack: &str, e: &Epoch) {
let v = json(pack, "vectors.json");
let ds = &e.dataset;
let head: Vec<u32> = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect();
for (i, h) in head.iter().enumerate() {
assert_eq!(ds.word(i as u32), *h, "{pack}: dataset[{i}]");
}
let last_index = v["dataset_last_index"].as_u64().unwrap() as u32;
assert_eq!(last_index, ds.mask);
assert_eq!(ds.word(last_index), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]");
let samples = v["dataset_samples"].as_array().unwrap();
assert_eq!(samples.len(), 64);
let mut n = 0;
for s in samples {
let idx = s["index"].as_u64().unwrap() as u32;
assert_eq!(ds.word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]");
n += 1;
}
assert_eq!(n, 64);
}
#[test]
fn dataset_words_match_memhard_pack() {
check_dataset_words("igneum-genesis-mh", mh_epoch());
}
#[test]
fn dataset_words_match_closed_packs() {
check_dataset_words("igneum-genesis", &closed_epoch("igneum-genesis"));
check_dataset_words("igneum-hourly", &closed_epoch("igneum-hourly"));
}
/// Returns the number of hashes compared (3 warps x 32 lanes = 96).
fn check_vectors(pack: &str, e: &Epoch) -> usize {
let v = json(pack, "vectors.json");
assert_eq!(v["dataset_mode"].as_str().unwrap(), e.dataset.mode().name());
assert_eq!(v["dataset_log2_words"].as_u64().unwrap() as u32, e.dataset.log2_words);
let warps = v["warps"].as_array().unwrap();
assert_eq!(warps.len(), 3);
let mut n = 0;
for w in warps {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let expected: Vec<u64> = w["expected"].as_array().unwrap().iter().map(hex64).collect();
let got = e.hash_warp(base);
for lane in 0..32 {
assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}");
n += 1;
}
// The single-nonce API agrees with the warp.
assert_eq!(e.hash(base + 7), expected[7]);
assert!(e.verify_block(base + 7, expected[7]));
assert!(!e.verify_block(base + 7, expected[7] - 1));
}
n
}
#[test]
fn vectors_memhard_96() {
assert_eq!(check_vectors("igneum-genesis-mh", mh_epoch()), 96);
}
#[test]
fn vectors_closed_form_genesis_96() {
assert_eq!(check_vectors("igneum-genesis", &closed_epoch("igneum-genesis")), 96);
}
#[test]
fn vectors_closed_form_hourly_96() {
assert_eq!(check_vectors("igneum-hourly", &closed_epoch("igneum-hourly")), 96);
}
fn assert_same_text(pack: &str, file: &str, got: &str) {
let want = read(pack, file);
if got != want {
// Find the first differing line for a readable failure.
let (gl, wl): (Vec<&str>, Vec<&str>) = (got.lines().collect(), want.lines().collect());
for i in 0..gl.len().max(wl.len()) {
let g = gl.get(i).copied().unwrap_or("<eof>");
let w = wl.get(i).copied().unwrap_or("<eof>");
if g != w {
panic!("{pack}/{file} differs at line {}:\n pack: {w}\n rust: {g}", i + 1);
}
}
panic!("{pack}/{file} differs only in trailing bytes (len {} vs {})", got.len(), want.len());
}
}
fn check_sources(pack: &str, e: &Epoch) {
let p = &e.program;
let mp = e.dataset.memhard().map(|m| &m.params);
assert_same_text(pack, "kernel.cu", &cuda_kernel(p, mp));
assert_same_text(pack, "program.metal", &metal_program(p, e.dataset.log2_words, LoadSource::Stored));
assert_same_text(pack, "kernel.cl", &opencl_kernel(p, mp));
assert_same_text(pack, "program.h", &program_header(p, DAY, &e.dataset.key, e.dataset.log2_words, mp));
if let Some(mp) = mp {
assert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp));
assert_same_text(pack, "memhard.metal", &metal_memhard(mp));
}
// program.json: byte-identical after normalising the Swift quoting defect in the "item" line.
let want = fix_swift_item_line(&read(pack, "program.json"));
let got = program_json(p, DAY, &e.dataset.key, e.dataset.log2_words, mp);
assert_eq!(got, want, "{pack}/program.json");
let _: Value = serde_json::from_str(&got).expect("Rust program.json is valid JSON");
}
#[test]
fn emitted_sources_match_memhard_pack() {
check_sources("igneum-genesis-mh", mh_epoch());
}
#[test]
fn emitted_sources_match_closed_packs() {
check_sources("igneum-genesis", &closed_epoch("igneum-genesis"));
check_sources("igneum-hourly", &closed_epoch("igneum-hourly"));
}
/// The whole pack as `export` writes it: vectors.json and vectors.h match the Swift ones apart from the
/// provenance string (the Swift adds "Metal GPU cross-check PASS", which the Rust side cannot claim).
fn check_export(pack: &str, e: &Epoch) {
let swift_v = json(pack, "vectors.json");
let source = swift_v["source"].as_str().unwrap();
let out = export_pack(e, DAY, source);
let file = |name: &str| -> &str { &out.files.iter().find(|(n, _)| n == name).unwrap().1 };
assert_same_text(pack, "vectors.json", file("vectors.json"));
assert_same_text(pack, "vectors.h", file("vectors.h"));
let expected: Vec<&str> = if e.dataset.mode() == DatasetMode::MemoryHard {
vec![
"program.json",
"vectors.json",
"kernel.cu",
"kernel.cl",
"program.h",
"vectors.h",
"program.metal",
"memhard.h",
"memhard.metal",
]
} else {
vec!["program.json", "vectors.json", "kernel.cu", "kernel.cl", "program.h", "vectors.h", "program.metal"]
};
assert_eq!(out.files.iter().map(|(n, _)| n.as_str()).collect::<Vec<_>>(), expected);
}
#[test]
fn export_pack_matches_memhard_pack() {
check_export("igneum-genesis-mh", mh_epoch());
}
#[test]
fn export_pack_matches_closed_packs() {
check_export("igneum-genesis", &closed_epoch("igneum-genesis"));
check_export("igneum-hourly", &closed_epoch("igneum-hourly"));
}
#[test]
fn item_is_independent_of_dataset_size() {
// MEMHARD.md 1.7: a smaller dataset is a prefix of items, so dataset[w] is the same at every size.
let big = &mh_epoch().dataset;
let m = big.memhard().unwrap();
for w in [0u32, 1, 15, 16, 17, 0x00ff_ffff, 0x03ff_ffff] {
assert_eq!(big.word(w), m.word(w));
}
}