igneum-pow: Rust crate bit-exact with proto-metal (seed, generator, memhard, verifier, emitters)
Standard-library Rust port of the Swift prototype for the rusty-kaspa fork. 23 tests tie it to proto-cuda/packs: program.json instruction by instruction for three packs, cache FNV 48c4f5bf24166b2e, dataset head/last/64 samples, 96/96 hash vectors per pack, and kernel.cu, program.metal, kernel.cl, program.h, memhard.h, memhard.metal byte-identical. CPU verify 0.41 to 0.58 ms per warp (Swift 0.63 to 1.21), cache fill 175 to 181 ms one core. CLI: bench, export, hash. program.json is written as valid JSON (the Swift quotes the cache line mask inside the "item" string; fix pending in main.swift). Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
401f6af1df
commit
ab99b67d3d
13 changed files with 2803 additions and 0 deletions
1
igneum-pow/.gitignore
vendored
Normal file
1
igneum-pow/.gitignore
vendored
Normal file
|
|
@ -0,0 +1 @@
|
|||
target/
|
||||
105
igneum-pow/Cargo.lock
generated
Normal file
105
igneum-pow/Cargo.lock
generated
Normal file
|
|
@ -0,0 +1,105 @@
|
|||
# This file is automatically @generated by Cargo.
|
||||
# It is not intended for manual editing.
|
||||
version = 4
|
||||
|
||||
[[package]]
|
||||
name = "igneum-pow"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"serde_json",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "itoa"
|
||||
version = "1.0.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
|
||||
|
||||
[[package]]
|
||||
name = "memchr"
|
||||
version = "2.8.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
|
||||
|
||||
[[package]]
|
||||
name = "proc-macro2"
|
||||
version = "1.0.107"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
|
||||
dependencies = [
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quote"
|
||||
version = "1.0.47"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
|
||||
dependencies = [
|
||||
"serde_core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_core"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
|
||||
dependencies = [
|
||||
"serde_derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_derive"
|
||||
version = "1.0.229"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_json"
|
||||
version = "1.0.151"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
|
||||
dependencies = [
|
||||
"itoa",
|
||||
"memchr",
|
||||
"serde",
|
||||
"serde_core",
|
||||
"zmij",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "3.0.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "unicode-ident"
|
||||
version = "1.0.26"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
|
||||
|
||||
[[package]]
|
||||
name = "zmij"
|
||||
version = "1.0.23"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
|
||||
30
igneum-pow/Cargo.toml
Normal file
30
igneum-pow/Cargo.toml
Normal file
|
|
@ -0,0 +1,30 @@
|
|||
[package]
|
||||
name = "igneum-pow"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
description = "Igneum random-program GPU proof-of-work: seed, program generator, memory-hard dataset, CPU warp verifier and kernel emitters, bit-exact with proto-metal"
|
||||
license = "MIT"
|
||||
publish = false
|
||||
|
||||
[lib]
|
||||
name = "igneum_pow"
|
||||
path = "src/lib.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "igneum-pow"
|
||||
path = "src/main.rs"
|
||||
|
||||
[dependencies]
|
||||
|
||||
[dev-dependencies]
|
||||
serde_json = "1"
|
||||
|
||||
# The cache fill is 2^22 ChaCha12 blocks and the vector tests derive thousands of items.
|
||||
# Unoptimised builds would make `cargo test` take minutes, so the dev profile is optimised too.
|
||||
[profile.dev]
|
||||
opt-level = 3
|
||||
|
||||
[profile.release]
|
||||
opt-level = 3
|
||||
lto = true
|
||||
codegen-units = 1
|
||||
97
igneum-pow/README.md
Normal file
97
igneum-pow/README.md
Normal file
|
|
@ -0,0 +1,97 @@
|
|||
# igneum-pow
|
||||
|
||||
The Igneum lottery hash in Rust, bit-exact with the Swift prototype in `proto-metal/main.swift`. This is the
|
||||
crate the rusty-kaspa fork will call (`docs/fork-map.md`, rows a1 to a3) so a node written in Rust can verify any
|
||||
block and hand miners the kernel source for the epoch. No dependency outside the standard library; `serde_json`
|
||||
is a dev-dependency for reading the packs in the tests.
|
||||
|
||||
Date: 3 October 2026. Toolchain: rustc 1.99.0 via rustup (the Homebrew 1.69 on PATH is too old; use
|
||||
`~/.cargo/bin/cargo`).
|
||||
|
||||
## Modules
|
||||
|
||||
| Module | What it is | Swift namesake |
|
||||
|---|---|---|
|
||||
| `seed` | 32-byte seed words from a string (FNV-1a 64, four salts, finalised); `seed_words_from_bytes` is the boundary where the chain will feed the VDF output; SplitMix64 | `seedWords`, `SplitMix64` |
|
||||
| `generator` | the 64-instruction program for a seed (op, dst, src, src2, imm, imm2, rot, bit, mask); levers `load_weight` and `wide_frac` | `generateProgram`, `GeneratorConfig` |
|
||||
| `memhard` | 256 MiB cache (2^16 chains of 64 ChaCha12 blocks), mixer parameters, 8-round item derivation with the 32 lanes interleaved, `MemhardCpu::fetch` | `cpuFillCache`, `MixParams`, `deriveItems`, `MemhardCPU` |
|
||||
| `verify` | the 32-lane warp interpreter, `DatasetMode::{ClosedForm, MemoryHard}`, `Epoch`, `hash_warp`, `verify_block` | `cpuWarp`, `DatasetSource` |
|
||||
| `emit` | Metal, CUDA and OpenCL source, program.h, memhard.h, vectors.h, program.json, vectors.json, `export_pack` | `generateMSL`, `memhardMSL`, `emitMemhardCore`, `generateCUDA`, `generateOpenCL`, `exportPack` |
|
||||
|
||||
## The API the fork calls
|
||||
|
||||
```rust
|
||||
use igneum_pow::{Epoch, DatasetMode};
|
||||
|
||||
// Once per epoch and day: generates the program and fills the 256 MiB cache (about 0.18 s on one core).
|
||||
let epoch = Epoch::memory_hard("igneum-genesis", "2026-10-03");
|
||||
|
||||
let h: u64 = epoch.hash(nonce); // one nonce (computes its aligned 32-nonce warp)
|
||||
let w: [u64; 32] = epoch.hash_warp(base_nonce); // one warp
|
||||
let ok: bool = epoch.verify_block(nonce, target_u64);
|
||||
|
||||
// Miner programs for the epoch, byte-identical to the Swift exporter.
|
||||
let pack = igneum_pow::emit::export_pack(&epoch, "2026-10-03", "igneum node");
|
||||
pack.write_to(std::path::Path::new("out"))?; // kernel.cu, kernel.cl, program.metal, memhard.h, ...
|
||||
```
|
||||
|
||||
`Epoch` is `Send + Sync`; build one and share it. `DatasetMode::ClosedForm` reproduces the two old packs
|
||||
(`igneum-genesis`, `igneum-hourly`) and is not memory-hard. The hash is 64 bits; the fork maps it into its
|
||||
256-bit target space in `consensus/pow/src/lib.rs`.
|
||||
|
||||
## CLI
|
||||
|
||||
```
|
||||
cargo build --release
|
||||
./target/release/igneum-pow bench --seed igneum-genesis [--warps 20] [--closed-form] [--day 2026-10-03]
|
||||
./target/release/igneum-pow export --seed igneum-genesis --out <dir> [--closed-form]
|
||||
./target/release/igneum-pow hash --seed igneum-genesis --nonce 4103
|
||||
```
|
||||
|
||||
## Tests
|
||||
|
||||
`cargo test` (23 tests, 0.7 s after compile; the dev profile is optimised so the cache fill is quick):
|
||||
|
||||
| Check | Pack | Result |
|
||||
|---|---|---|
|
||||
| program.json instruction by instruction, op mix, loads per hash | igneum-genesis, igneum-genesis-mh, igneum-hourly | 3 x 64 match |
|
||||
| Mixer parameters (key, rot, mul, rc) | igneum-genesis-mh | match |
|
||||
| Cache head, last line, FNV-1a 64 `48c4f5bf24166b2e` | igneum-genesis-mh | match |
|
||||
| Dataset head (16), `[MASK]`, 64 sampled words | all three | match |
|
||||
| 96 hash vectors (3 warps x 32 lanes) | igneum-genesis-mh | 96/96 |
|
||||
| 96 hash vectors | igneum-genesis, igneum-hourly | 96/96 each |
|
||||
| kernel.cu, program.metal, kernel.cl, program.h byte-identical | all three | identical |
|
||||
| memhard.h, memhard.metal byte-identical | igneum-genesis-mh | identical |
|
||||
| program.json byte-identical (after the fix below) | all three | identical |
|
||||
| vectors.json, vectors.h byte-identical apart from the provenance string | all three | identical |
|
||||
|
||||
An independent `diff -r` of `igneum-pow export` output against the checked-in packs shows the same two lines
|
||||
only: the provenance string and the `"item"` line.
|
||||
|
||||
One deliberate difference: `proto-cuda/packs/igneum-genesis-mh/program.json` as written by the Swift is not
|
||||
valid JSON (main.swift line 1291 uses `jhex` inside the `"item"` string, so the cache line mask is quoted inside a
|
||||
quoted string). The Rust emitter writes `0x003fffff` bare; the test normalises that one line before comparing.
|
||||
A node must hand miners valid JSON, so the Rust side does not reproduce the defect.
|
||||
|
||||
## Measured, 3 October 2026, Apple M5 Max, one core, release build
|
||||
|
||||
| Step | Rust | Swift (MEMHARD.md) |
|
||||
|---|---|---|
|
||||
| Cache fill, 256 MiB, 65,536 chains x 64 ChaCha12 blocks | 175 to 181 ms (5 quiet runs; 200 ms once with another build running) | 184.5 to 190.6 ms (C++ host reference 161.5) |
|
||||
| CPU verify per warp, igneum-genesis, 104 loads, 3,328 items, avg of 20 | 0.441 ms | 0.649 ms |
|
||||
| igneum-genesis/epoch1, 104 loads | 0.411 ms | 0.631 ms |
|
||||
| igneum-genesis/epoch2, 112 loads | 0.488 ms | 0.701 ms |
|
||||
| igneum-second-seed, 104 loads | 0.482 ms | 0.801 ms |
|
||||
| igneum-second-seed/epoch1, 144 loads, 4,608 items | 0.579 ms | 1.205 ms |
|
||||
| Cold single warps across the five seeds | 0.41 to 0.87 ms | 1.16 to 2.11 ms |
|
||||
| Closed form, igneum-genesis | 0.002 ms | 0.017 ms |
|
||||
|
||||
The Rust verifier is 1.4x to 2.1x faster than the Swift one per warp; the registers are kept register-major
|
||||
(`r[reg][lane]`) so the lane loops vectorise, and the item derivation interleaves the 32 lanes round by round as
|
||||
the Swift does. The 10 ms gate holds with a margin of about 17x on the steady figure and 11x on the worst cold warp.
|
||||
|
||||
## Not done here
|
||||
|
||||
- No GPU. The vectors tie this crate to the Metal and CUDA results through the packs; nothing here runs a kernel.
|
||||
- The epoch seed is still a string. `seed::seed_words_from_bytes` is where the VDF output will enter.
|
||||
- The 256-bit target mapping and the `kaspa_pow::State` shape belong to the fork, not to this crate.
|
||||
2
igneum-pow/rustfmt.toml
Normal file
2
igneum-pow/rustfmt.toml
Normal file
|
|
@ -0,0 +1,2 @@
|
|||
max_width = 120
|
||||
use_small_heuristics = "Max"
|
||||
1044
igneum-pow/src/emit.rs
Normal file
1044
igneum-pow/src/emit.rs
Normal file
File diff suppressed because it is too large
Load diff
276
igneum-pow/src/generator.rs
Normal file
276
igneum-pow/src/generator.rs
Normal file
|
|
@ -0,0 +1,276 @@
|
|||
//! The program generator: 64 integer instructions over 8 x u32 lane registers, run for 8 iterations.
|
||||
//! Draw order, weights and the lever rules are those of `generateProgram` in `proto-metal/main.swift`.
|
||||
|
||||
use crate::seed::{program_rng, seed_words};
|
||||
|
||||
/// Iterations of the instruction list per hash.
|
||||
pub const ITERATIONS: usize = 8;
|
||||
/// Instructions per program.
|
||||
pub const INSTR_COUNT: usize = 64;
|
||||
/// Lanes per verification unit (one SIMD group / warp).
|
||||
pub const LANES: usize = 32;
|
||||
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
|
||||
pub enum Op {
|
||||
Add,
|
||||
Sub,
|
||||
Mul,
|
||||
MulHi,
|
||||
Xor,
|
||||
Or,
|
||||
Rotl,
|
||||
Rotr,
|
||||
Mad,
|
||||
Shfl,
|
||||
Load,
|
||||
/// Warp-coalesced load (lever b). Never emitted unless `wide_frac > 0`.
|
||||
WLoad,
|
||||
}
|
||||
|
||||
impl Op {
|
||||
/// The name used in program.json, kernel comments and the op mix.
|
||||
pub fn name(self) -> &'static str {
|
||||
match self {
|
||||
Op::Add => "add",
|
||||
Op::Sub => "sub",
|
||||
Op::Mul => "mul",
|
||||
Op::MulHi => "mulhi",
|
||||
Op::Xor => "xor",
|
||||
Op::Or => "or",
|
||||
Op::Rotl => "rotl",
|
||||
Op::Rotr => "rotr",
|
||||
Op::Mad => "mad",
|
||||
Op::Shfl => "shfl",
|
||||
Op::Load => "load",
|
||||
Op::WLoad => "wload",
|
||||
}
|
||||
}
|
||||
|
||||
pub fn from_name(s: &str) -> Option<Op> {
|
||||
Some(match s {
|
||||
"add" => Op::Add,
|
||||
"sub" => Op::Sub,
|
||||
"mul" => Op::Mul,
|
||||
"mulhi" => Op::MulHi,
|
||||
"xor" => Op::Xor,
|
||||
"or" => Op::Or,
|
||||
"rotl" => Op::Rotl,
|
||||
"rotr" => Op::Rotr,
|
||||
"mad" => Op::Mad,
|
||||
"shfl" => Op::Shfl,
|
||||
"load" => Op::Load,
|
||||
"wload" => Op::WLoad,
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// One instruction. Every field is drawn for every instruction whether the op uses it or not, so the
|
||||
/// draw stream is identical for every op.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub struct Instr {
|
||||
pub op: Op,
|
||||
/// Destination register 0..7.
|
||||
pub dst: u8,
|
||||
/// Source register 0..7, never equal to `dst`.
|
||||
pub src: u8,
|
||||
/// Second source (mad only).
|
||||
pub src2: u8,
|
||||
/// Add immediate A.
|
||||
pub imm: u32,
|
||||
/// Add immediate B.
|
||||
pub imm2: u32,
|
||||
/// rotl amount 1..31.
|
||||
pub rot: u32,
|
||||
/// Selector bit of r0 for add, 0..31.
|
||||
pub bit: u8,
|
||||
/// Shuffle xor mask: 1, 2, 4, 8 or 16.
|
||||
pub mask: u8,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct Program {
|
||||
pub seed_string: String,
|
||||
pub seed: [u32; 8],
|
||||
pub instrs: Vec<Instr>,
|
||||
}
|
||||
|
||||
impl Program {
|
||||
pub fn loads_per_hash(&self) -> usize {
|
||||
self.instrs.iter().filter(|i| i.op == Op::Load || i.op == Op::WLoad).count() * ITERATIONS
|
||||
}
|
||||
pub fn wide_loads_per_hash(&self) -> usize {
|
||||
self.instrs.iter().filter(|i| i.op == Op::WLoad).count() * ITERATIONS
|
||||
}
|
||||
pub fn has_wide(&self) -> bool {
|
||||
self.instrs.iter().any(|i| i.op == Op::WLoad)
|
||||
}
|
||||
/// Distinct dataset items a 32-lane warp touches per hash: 32 per plain load, 2 per wide load.
|
||||
pub fn items_per_warp(&self) -> usize {
|
||||
(self.loads_per_hash() - self.wide_loads_per_hash()) * 32 + self.wide_loads_per_hash() * 2
|
||||
}
|
||||
/// Op histogram, count descending then name ascending, as the Swift prints it.
|
||||
pub fn histogram(&self) -> Vec<(&'static str, usize)> {
|
||||
let mut counts: Vec<(&'static str, usize)> = Vec::new();
|
||||
for i in &self.instrs {
|
||||
let name = i.op.name();
|
||||
match counts.iter_mut().find(|(n, _)| *n == name) {
|
||||
Some(e) => e.1 += 1,
|
||||
None => counts.push((name, 1)),
|
||||
}
|
||||
}
|
||||
counts.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(b.0)));
|
||||
counts
|
||||
}
|
||||
/// "load=13 xor=13 ..." as written into program.h.
|
||||
pub fn op_mix(&self) -> String {
|
||||
self.histogram().iter().map(|(n, c)| format!("{n}={c}")).collect::<Vec<_>>().join(" ")
|
||||
}
|
||||
}
|
||||
|
||||
/// Weights sum to 100. Loads are 25 percent so the kernel leans on memory.
|
||||
pub const OP_WEIGHTS: [(Op, u64); 11] = [
|
||||
(Op::Load, 25),
|
||||
(Op::Add, 12),
|
||||
(Op::Xor, 10),
|
||||
(Op::Mul, 8),
|
||||
(Op::Mad, 8),
|
||||
(Op::Shfl, 8),
|
||||
(Op::Rotl, 7),
|
||||
(Op::Sub, 6),
|
||||
(Op::MulHi, 6),
|
||||
(Op::Rotr, 6),
|
||||
(Op::Or, 4),
|
||||
];
|
||||
|
||||
/// Generator levers (MEMHARD.md section 2.4). The defaults reproduce the original generator exactly.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub struct GeneratorConfig {
|
||||
/// Percent weight of the load op.
|
||||
pub load_weight: u64,
|
||||
/// Percent of load instructions emitted as warp-coalesced wide loads.
|
||||
pub wide_frac: u64,
|
||||
}
|
||||
|
||||
impl Default for GeneratorConfig {
|
||||
fn default() -> Self {
|
||||
Self { load_weight: 25, wide_frac: 0 }
|
||||
}
|
||||
}
|
||||
|
||||
impl GeneratorConfig {
|
||||
/// Scaled weights: load gets `load_weight`, the other ten ops share the rest in their original
|
||||
/// proportions, rounded by largest remainder so the table still sums to 100.
|
||||
pub fn weights(&self) -> Vec<(Op, u64)> {
|
||||
if self.load_weight == 25 {
|
||||
return OP_WEIGHTS.to_vec();
|
||||
}
|
||||
let others = &OP_WEIGHTS[1..];
|
||||
let total: u64 = others.iter().map(|w| w.1).sum(); // 75
|
||||
let budget = 100 - self.load_weight;
|
||||
let mut scaled: Vec<(Op, u64, u64)> =
|
||||
others.iter().map(|&(op, w)| (op, (w * budget) / total, (w * budget) % total)).collect();
|
||||
let mut sum: u64 = scaled.iter().map(|s| s.1).sum();
|
||||
let mut order: Vec<usize> = (0..scaled.len()).collect();
|
||||
order.sort_by(|&a, &b| scaled[b].2.cmp(&scaled[a].2).then_with(|| a.cmp(&b)));
|
||||
let mut k = 0;
|
||||
while sum < budget {
|
||||
scaled[order[k]].1 += 1;
|
||||
sum += 1;
|
||||
k += 1;
|
||||
}
|
||||
let mut out = vec![(Op::Load, self.load_weight)];
|
||||
out.extend(scaled.iter().map(|s| (s.0, s.1)));
|
||||
out
|
||||
}
|
||||
}
|
||||
|
||||
/// The default generator for a seed string.
|
||||
pub fn generate(seed_string: &str) -> Program {
|
||||
generate_with(seed_string, &GeneratorConfig::default())
|
||||
}
|
||||
|
||||
/// The generator with levers. `generateProgram` in the Swift, draw for draw.
|
||||
pub fn generate_with(seed_string: &str, cfg: &GeneratorConfig) -> Program {
|
||||
let seed = seed_words(seed_string);
|
||||
generate_from_words(seed_string, seed, cfg)
|
||||
}
|
||||
|
||||
/// The generator from already-derived seed words (what the chain will call once the VDF output is in).
|
||||
pub fn generate_from_words(seed_string: &str, seed: [u32; 8], cfg: &GeneratorConfig) -> Program {
|
||||
let mut rng = program_rng(&seed);
|
||||
let weights = cfg.weights();
|
||||
let mut instrs = Vec::with_capacity(INSTR_COUNT);
|
||||
for _ in 0..INSTR_COUNT {
|
||||
let mut roll = rng.below(100);
|
||||
let mut op = Op::Add;
|
||||
for &(o, w) in &weights {
|
||||
if roll < w {
|
||||
op = o;
|
||||
break;
|
||||
}
|
||||
roll -= w;
|
||||
}
|
||||
let dst = rng.below(8);
|
||||
let mut a = rng.below(7);
|
||||
if a >= dst {
|
||||
a += 1;
|
||||
}
|
||||
let b = rng.below(8);
|
||||
let imm = rng.next() as u32;
|
||||
let imm2 = rng.next() as u32;
|
||||
let rot = 1 + rng.below(31) as u32;
|
||||
let bit = rng.below(32);
|
||||
let mask = 1u8 << rng.below(5);
|
||||
// Lever (b): the already-drawn selector bit decides whether a load is wide, so the stream is unchanged.
|
||||
if op == Op::Load && bit * 100 < cfg.wide_frac * 32 {
|
||||
op = Op::WLoad;
|
||||
}
|
||||
instrs.push(Instr { op, dst: dst as u8, src: a as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask });
|
||||
}
|
||||
Program { seed_string: seed_string.to_string(), seed, instrs }
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn default_weights_unchanged() {
|
||||
assert_eq!(GeneratorConfig::default().weights(), OP_WEIGHTS.to_vec());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn load_weight_17_table() {
|
||||
// MEMHARD.md section 2.4: add=13 xor=11 mul=9 mad=9 shfl=9 rotl=8 sub=7 mulhi=7 rotr=6 or=4.
|
||||
let w = GeneratorConfig { load_weight: 17, wide_frac: 0 }.weights();
|
||||
let expect = [
|
||||
(Op::Load, 17),
|
||||
(Op::Add, 13),
|
||||
(Op::Xor, 11),
|
||||
(Op::Mul, 9),
|
||||
(Op::Mad, 9),
|
||||
(Op::Shfl, 9),
|
||||
(Op::Rotl, 8),
|
||||
(Op::Sub, 7),
|
||||
(Op::MulHi, 7),
|
||||
(Op::Rotr, 6),
|
||||
(Op::Or, 4),
|
||||
];
|
||||
assert_eq!(w, expect.to_vec());
|
||||
assert_eq!(w.iter().map(|x| x.1).sum::<u64>(), 100);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn genesis_shape() {
|
||||
let p = generate("igneum-genesis");
|
||||
assert_eq!(p.instrs.len(), 64);
|
||||
assert_eq!(p.loads_per_hash(), 104);
|
||||
assert_eq!(p.op_mix(), "load=13 xor=13 sub=7 shfl=6 add=5 mulhi=5 mad=4 rotr=4 mul=3 rotl=3 or=1");
|
||||
for i in &p.instrs {
|
||||
assert_ne!(i.dst, i.src);
|
||||
assert!((1..=31).contains(&i.rot));
|
||||
assert!(i.mask.is_power_of_two() && i.mask <= 16);
|
||||
}
|
||||
}
|
||||
}
|
||||
30
igneum-pow/src/lib.rs
Normal file
30
igneum-pow/src/lib.rs
Normal file
|
|
@ -0,0 +1,30 @@
|
|||
//! igneum-pow: the Igneum lottery hash, bit-exact with the Swift prototype in `proto-metal/main.swift`.
|
||||
//!
|
||||
//! The crate has five parts, each mirroring one section of the prototype:
|
||||
//!
|
||||
//! * [`seed`]: the 32-byte seed words from a string (FNV-1a 64, four salts) and the SplitMix64 stream.
|
||||
//! * [`generator`]: the 64-instruction program drawn from a seed.
|
||||
//! * [`memhard`]: the 256 MiB ChaCha12 cache and the 8-round dataset item derivation (`proto-metal/MEMHARD.md`).
|
||||
//! * [`verify`]: the 32-lane warp interpreter that computes the 64-bit hash on the CPU, deriving dataset
|
||||
//! words on demand from the cache (or from the closed form, for the old packs).
|
||||
//! * [`emit`]: the Metal, CUDA and OpenCL kernel text for a program, byte-identical to the Swift exporter.
|
||||
//!
|
||||
//! Nothing here depends on a crate outside the standard library. The integration points for the
|
||||
//! rusty-kaspa fork (`docs/fork-map.md`) are [`verify::Epoch`], [`verify::Epoch::verify_block`] and
|
||||
//! [`verify::Epoch::hash_warp`]; the node hands [`emit::Pack`] files to miners.
|
||||
|
||||
// The lane loops are written index style on purpose so they read like the kernels they mirror, and the
|
||||
// SplitMix64 `next` keeps the Swift name.
|
||||
#![allow(clippy::needless_range_loop, clippy::should_implement_trait, clippy::large_enum_variant)]
|
||||
#![allow(clippy::manual_slice_size_calculation, clippy::too_many_arguments)]
|
||||
|
||||
pub mod emit;
|
||||
pub mod generator;
|
||||
pub mod memhard;
|
||||
pub mod seed;
|
||||
pub mod verify;
|
||||
|
||||
pub use generator::{generate, Instr, Op, Program};
|
||||
pub use memhard::{Cache, MemhardCpu, MixParams};
|
||||
pub use seed::{fnv1a64, seed_words, SplitMix64};
|
||||
pub use verify::{hash_warp, verify_block, DatasetMode, DatasetSource, Epoch};
|
||||
156
igneum-pow/src/main.rs
Normal file
156
igneum-pow/src/main.rs
Normal file
|
|
@ -0,0 +1,156 @@
|
|||
//! igneum-pow CLI.
|
||||
//!
|
||||
//! igneum-pow bench --seed <s> [--day <d>] [--closed-form] [--dataset-log2 28] [--warps 20]
|
||||
//! igneum-pow export --seed <s> --out <dir> [--day <d>] [--closed-form] [--dataset-log2 28]
|
||||
//! igneum-pow hash --seed <s> --nonce <n> [--day <d>] [--closed-form] [--dataset-log2 28]
|
||||
|
||||
use igneum_pow::emit::export_pack;
|
||||
use igneum_pow::memhard::Cache;
|
||||
use igneum_pow::seed::day_key;
|
||||
use igneum_pow::verify::{DatasetMode, Epoch, DEFAULT_DATASET_LOG2};
|
||||
use std::time::Instant;
|
||||
|
||||
struct Args {
|
||||
cmd: String,
|
||||
seed: String,
|
||||
day: String,
|
||||
out: Option<String>,
|
||||
closed_form: bool,
|
||||
dataset_log2: u32,
|
||||
warps: usize,
|
||||
nonce: u32,
|
||||
}
|
||||
|
||||
fn usage() -> ! {
|
||||
eprintln!(
|
||||
"igneum-pow <bench|export|hash> --seed <string> [--day 2026-10-03] [--closed-form] [--dataset-log2 28]\n\
|
||||
\x20 bench [--warps 20] fill the cache, then time the CPU verifier per 32-lane warp\n\
|
||||
\x20 export --out <dir> write the program pack (kernel.cu, kernel.cl, program.metal, memhard.h, ...)\n\
|
||||
\x20 hash --nonce <n> print the 64-bit hash of one nonce"
|
||||
);
|
||||
std::process::exit(2)
|
||||
}
|
||||
|
||||
fn parse() -> Args {
|
||||
let mut a = Args {
|
||||
cmd: String::new(),
|
||||
seed: "igneum-genesis".into(),
|
||||
day: "2026-10-03".into(),
|
||||
out: None,
|
||||
closed_form: false,
|
||||
dataset_log2: DEFAULT_DATASET_LOG2,
|
||||
warps: 20,
|
||||
nonce: 0,
|
||||
};
|
||||
let mut it = std::env::args().skip(1);
|
||||
a.cmd = it.next().unwrap_or_else(|| usage());
|
||||
while let Some(k) = it.next() {
|
||||
let mut val = || it.next().unwrap_or_else(|| usage());
|
||||
match k.as_str() {
|
||||
"--seed" => a.seed = val(),
|
||||
"--day" => a.day = val(),
|
||||
"--out" => a.out = Some(val()),
|
||||
"--closed-form" => a.closed_form = true,
|
||||
"--dataset-log2" => a.dataset_log2 = val().parse().unwrap_or_else(|_| usage()),
|
||||
"--warps" => a.warps = val().parse().unwrap_or_else(|_| usage()),
|
||||
"--nonce" => a.nonce = val().parse().unwrap_or_else(|_| usage()),
|
||||
_ => usage(),
|
||||
}
|
||||
}
|
||||
a
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let a = parse();
|
||||
let mode = if a.closed_form { DatasetMode::ClosedForm } else { DatasetMode::MemoryHard };
|
||||
match a.cmd.as_str() {
|
||||
"bench" => bench(&a, mode),
|
||||
"export" => export(&a, mode),
|
||||
"hash" => {
|
||||
let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2);
|
||||
println!("{:016x}", e.hash(a.nonce));
|
||||
}
|
||||
_ => usage(),
|
||||
}
|
||||
}
|
||||
|
||||
fn bench(a: &Args, mode: DatasetMode) {
|
||||
println!(
|
||||
"igneum-pow bench: seed \"{}\", day \"{}\", dataset 2^{} words ({})",
|
||||
a.seed,
|
||||
a.day,
|
||||
a.dataset_log2,
|
||||
mode.name()
|
||||
);
|
||||
if mode == DatasetMode::MemoryHard {
|
||||
// Time the cache fill on its own first (one core), then build the epoch (which fills it again).
|
||||
let t0 = Instant::now();
|
||||
let c = Cache::fill(day_key(&a.day));
|
||||
let fill_ms = t0.elapsed().as_secs_f64() * 1e3;
|
||||
println!("cache: fill {fill_ms:.1} ms on one core (2^26 words, 65536 chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", c.fnv1a64());
|
||||
drop(c);
|
||||
}
|
||||
let t0 = Instant::now();
|
||||
let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2);
|
||||
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
|
||||
println!(
|
||||
"program: {} loads/hash, {} items/warp, op mix {}; epoch built in {build_ms:.1} ms",
|
||||
e.program.loads_per_hash(),
|
||||
e.program.items_per_warp(),
|
||||
e.program.op_mix()
|
||||
);
|
||||
let bases = [0u32, 4096, 1_000_000];
|
||||
for &b in &bases {
|
||||
let t = Instant::now();
|
||||
let r = e.interpret_warp(b);
|
||||
let ms = t.elapsed().as_secs_f64() * 1e3;
|
||||
println!(
|
||||
"warp base {b}: single cold run {ms:.3} ms, {} items derived, lane0 {:016x} lane31 {:016x}",
|
||||
r.items_derived, r.hashes[0], r.hashes[31]
|
||||
);
|
||||
}
|
||||
let n = a.warps.max(1);
|
||||
let t = Instant::now();
|
||||
let mut sink = 0u64;
|
||||
for i in 0..n {
|
||||
let w = e.hash_warp((i as u32) * 32 + 65536);
|
||||
sink ^= w[0];
|
||||
}
|
||||
let avg = t.elapsed().as_secs_f64() * 1e3 / n as f64;
|
||||
println!("CPU verify: {avg:.3} ms per 32-lane warp, avg of {n} (checksum {sink:016x})");
|
||||
}
|
||||
|
||||
fn export(a: &Args, mode: DatasetMode) {
|
||||
let out = a.out.clone().unwrap_or_else(|| usage());
|
||||
let t0 = Instant::now();
|
||||
let e = Epoch::new(&a.seed, &a.day, mode, a.dataset_log2);
|
||||
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
|
||||
println!("igneum-pow export {out}");
|
||||
println!(
|
||||
"seed \"{}\", day \"{}\", dataset 2^{} words ({}), loads/hash {}, wide loads/hash {}; epoch built in {build_ms:.1} ms",
|
||||
a.seed,
|
||||
a.day,
|
||||
a.dataset_log2,
|
||||
mode.name(),
|
||||
e.program.loads_per_hash(),
|
||||
e.program.wide_loads_per_hash()
|
||||
);
|
||||
println!("op mix: {}", e.program.op_mix());
|
||||
let source = format!("igneum-pow (Rust) CPU interpreter, {} dataset", mode.name());
|
||||
let pack = export_pack(&e, &a.day, &source);
|
||||
let dir = std::path::Path::new(&out);
|
||||
if let Err(err) = pack.write_to(dir) {
|
||||
eprintln!("FAIL: write error {err}");
|
||||
std::process::exit(1);
|
||||
}
|
||||
for (name, text) in &pack.files {
|
||||
println!("wrote {}/{name} ({} bytes)", dir.display(), text.len());
|
||||
}
|
||||
for (i, b) in pack.bases.iter().enumerate() {
|
||||
println!("vector warp base {b}: lane0 {:016x} lane31 {:016x}", pack.outs[i][0], pack.outs[i][31]);
|
||||
}
|
||||
if mode == DatasetMode::MemoryHard {
|
||||
println!("cache FNV-1a 64 {:016x}", pack.vectors.cache_fnv);
|
||||
}
|
||||
println!("OVERALL: PASS (pack written)");
|
||||
}
|
||||
313
igneum-pow/src/memhard.rs
Normal file
313
igneum-pow/src/memhard.rs
Normal file
|
|
@ -0,0 +1,313 @@
|
|||
//! The memory-hard dataset of `proto-metal/MEMHARD.md`: a 256 MiB cache of chained ChaCha12 blocks keyed
|
||||
//! by the day key, and 64-byte dataset items derived by 8 dependent cache reads through a seed-parameterised
|
||||
//! ARX-multiply mixer. The verifier holds the cache and never the dataset.
|
||||
//!
|
||||
//! All arithmetic is on u32 modulo 2^32. Rotations are by 1..31 at every call site.
|
||||
|
||||
use crate::seed::{day_key, fnv1a64_words, SplitMix64};
|
||||
|
||||
pub const CACHE_LOG2_WORDS: usize = 26;
|
||||
pub const CACHE_SEGMENT_LOG2_LINES: usize = 6;
|
||||
/// 2^26 words = 256 MiB.
|
||||
pub const CACHE_WORDS: usize = 1 << CACHE_LOG2_WORDS;
|
||||
/// 2^22 lines of 16 words.
|
||||
pub const CACHE_LINES: usize = CACHE_WORDS >> 4;
|
||||
/// 64 chained lines per segment.
|
||||
pub const CACHE_LINES_PER_SEGMENT: usize = 1 << CACHE_SEGMENT_LOG2_LINES;
|
||||
/// 2^16 independent segments.
|
||||
pub const CACHE_SEGMENTS: usize = CACHE_LINES >> CACHE_SEGMENT_LOG2_LINES;
|
||||
pub const CACHE_LINE_MASK: u32 = (CACHE_LINES - 1) as u32;
|
||||
pub const ITEM_ROUNDS: usize = 8;
|
||||
pub const CHACHA_ROUNDS: usize = 12;
|
||||
/// The ChaCha constants "expand 32-byte k".
|
||||
pub const CHACHA_SIGMA: [u32; 4] = [0x61707865, 0x3320646e, 0x79622d32, 0x6b206574];
|
||||
/// "Igne", "umMH".
|
||||
pub const CACHE_TAG: [u32; 2] = [0x49676e65, 0x756d4d48];
|
||||
|
||||
#[inline(always)]
|
||||
fn rotl(x: u32, n: u32) -> u32 {
|
||||
x.rotate_left(n)
|
||||
}
|
||||
|
||||
/// The ChaCha quarter round with explicit rotations.
|
||||
#[inline(always)]
|
||||
fn qr(s: &mut [u32; 16], a: usize, b: usize, c: usize, d: usize, r1: u32, r2: u32, r3: u32, r4: u32) {
|
||||
s[a] = s[a].wrapping_add(s[b]);
|
||||
s[d] ^= s[a];
|
||||
s[d] = rotl(s[d], r1);
|
||||
s[c] = s[c].wrapping_add(s[d]);
|
||||
s[b] ^= s[c];
|
||||
s[b] = rotl(s[b], r2);
|
||||
s[a] = s[a].wrapping_add(s[b]);
|
||||
s[d] ^= s[a];
|
||||
s[d] = rotl(s[d], r3);
|
||||
s[c] = s[c].wrapping_add(s[d]);
|
||||
s[b] ^= s[c];
|
||||
s[b] = rotl(s[b], r4);
|
||||
}
|
||||
|
||||
/// `y = ChaCha12 core(x) + x`. Standard rotations 16, 12, 8, 7; column round then diagonal round, six times.
|
||||
#[inline]
|
||||
pub fn chacha_block(x: &[u32; 16]) -> [u32; 16] {
|
||||
let mut y = *x;
|
||||
for _ in 0..CHACHA_ROUNDS / 2 {
|
||||
qr(&mut y, 0, 4, 8, 12, 16, 12, 8, 7);
|
||||
qr(&mut y, 1, 5, 9, 13, 16, 12, 8, 7);
|
||||
qr(&mut y, 2, 6, 10, 14, 16, 12, 8, 7);
|
||||
qr(&mut y, 3, 7, 11, 15, 16, 12, 8, 7);
|
||||
qr(&mut y, 0, 5, 10, 15, 16, 12, 8, 7);
|
||||
qr(&mut y, 1, 6, 11, 12, 16, 12, 8, 7);
|
||||
qr(&mut y, 2, 7, 8, 13, 16, 12, 8, 7);
|
||||
qr(&mut y, 3, 4, 9, 14, 16, 12, 8, 7);
|
||||
}
|
||||
for i in 0..16 {
|
||||
y[i] = y[i].wrapping_add(x[i]);
|
||||
}
|
||||
y
|
||||
}
|
||||
|
||||
/// Mixer parameters drawn from the day key. Draw order: ROT[0..7] (1..31), MUL[0..15] (odd), RC[0..15].
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct MixParams {
|
||||
pub key: [u32; 8],
|
||||
pub rot: [u32; 8],
|
||||
pub mul: [u32; 16],
|
||||
pub rc: [u32; 16],
|
||||
}
|
||||
|
||||
impl MixParams {
|
||||
pub fn new(key: [u32; 8]) -> Self {
|
||||
let mut rng = SplitMix64::new(key[0] as u64 | ((key[1] as u64) << 32));
|
||||
let mut rot = [0u32; 8];
|
||||
let mut mul = [0u32; 16];
|
||||
let mut rc = [0u32; 16];
|
||||
for r in rot.iter_mut() {
|
||||
*r = 1 + rng.below(31) as u32;
|
||||
}
|
||||
for m in mul.iter_mut() {
|
||||
*m = (rng.next() as u32) | 1;
|
||||
}
|
||||
for c in rc.iter_mut() {
|
||||
*c = rng.next() as u32;
|
||||
}
|
||||
Self { key, rot, mul, rc }
|
||||
}
|
||||
/// Parameters for a day string: the key is `seed_words("day/" + day)`.
|
||||
pub fn for_day(day: &str) -> Self {
|
||||
Self::new(day_key(day))
|
||||
}
|
||||
}
|
||||
|
||||
/// Round key `(r + 1) * 0x9E3779B9` mod 2^32.
|
||||
#[inline(always)]
|
||||
pub fn round_key(r: usize) -> u32 {
|
||||
((r + 1) as u32).wrapping_mul(0x9E3779B9)
|
||||
}
|
||||
|
||||
/// `M_r` on 16 words in place: per word `(s ^ (RC + rk)) * MUL`, then one ChaCha-shaped double round with
|
||||
/// the four column rotations `ROT[0..3]` and the four diagonal rotations `ROT[4..7]`.
|
||||
#[inline(always)]
|
||||
pub fn mixer(s: &mut [u32; 16], rk: u32, mp: &MixParams) {
|
||||
for i in 0..16 {
|
||||
s[i] = (s[i] ^ mp.rc[i].wrapping_add(rk)).wrapping_mul(mp.mul[i]);
|
||||
}
|
||||
let r = &mp.rot;
|
||||
qr(s, 0, 4, 8, 12, r[0], r[1], r[2], r[3]);
|
||||
qr(s, 1, 5, 9, 13, r[0], r[1], r[2], r[3]);
|
||||
qr(s, 2, 6, 10, 14, r[0], r[1], r[2], r[3]);
|
||||
qr(s, 3, 7, 11, 15, r[0], r[1], r[2], r[3]);
|
||||
qr(s, 0, 5, 10, 15, r[4], r[5], r[6], r[7]);
|
||||
qr(s, 1, 6, 11, 12, r[4], r[5], r[6], r[7]);
|
||||
qr(s, 2, 7, 8, 13, r[4], r[5], r[6], r[7]);
|
||||
qr(s, 3, 4, 9, 14, r[4], r[5], r[6], r[7]);
|
||||
}
|
||||
|
||||
/// The 256 MiB cache for one day key.
|
||||
pub struct Cache {
|
||||
pub key: [u32; 8],
|
||||
words: Vec<u32>,
|
||||
}
|
||||
|
||||
impl Cache {
|
||||
/// One segment: 64 chained lines written at `cache[seg * 1024 ..]`.
|
||||
/// `in_j = prev XOR (sigma || K || seg || j || tag)`, `line_j = B(in_j)`, `prev_0 = 0`.
|
||||
pub fn fill_segment(words: &mut [u32], seg: usize, key: &[u32; 8]) {
|
||||
let base = (seg << CACHE_SEGMENT_LOG2_LINES) * 16;
|
||||
let seg_words = &mut words[base..base + CACHE_LINES_PER_SEGMENT * 16];
|
||||
let mut prev = [0u32; 16];
|
||||
for (j, line) in seg_words.as_chunks_mut::<16>().0.iter_mut().enumerate() {
|
||||
let mut x = [0u32; 16];
|
||||
x[..4].copy_from_slice(&CHACHA_SIGMA);
|
||||
x[4..12].copy_from_slice(key);
|
||||
x[12] = seg as u32;
|
||||
x[13] = j as u32;
|
||||
x[14] = CACHE_TAG[0];
|
||||
x[15] = CACHE_TAG[1];
|
||||
for i in 0..16 {
|
||||
x[i] ^= prev[i];
|
||||
}
|
||||
let y = chacha_block(&x);
|
||||
line.copy_from_slice(&y);
|
||||
prev = y;
|
||||
}
|
||||
}
|
||||
|
||||
/// The whole cache on the calling thread: 65,536 chains of 64 ChaCha12 blocks, in segment order.
|
||||
pub fn fill(key: [u32; 8]) -> Cache {
|
||||
let mut words = vec![0u32; CACHE_WORDS];
|
||||
for seg in 0..CACHE_SEGMENTS {
|
||||
Self::fill_segment(&mut words, seg, &key);
|
||||
}
|
||||
Cache { key, words }
|
||||
}
|
||||
|
||||
pub fn for_day(day: &str) -> Cache {
|
||||
Self::fill(day_key(day))
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn words(&self) -> &[u32] {
|
||||
&self.words
|
||||
}
|
||||
|
||||
/// Cache line `a` (0 <= a < 2^22) as 16 words.
|
||||
#[inline(always)]
|
||||
pub fn line(&self, a: u32) -> &[u32] {
|
||||
let o = (a & CACHE_LINE_MASK) as usize * 16;
|
||||
&self.words[o..o + 16]
|
||||
}
|
||||
|
||||
/// FNV-1a 64 over the cache as little-endian bytes (what `vectors.h` carries as `IGNEUM_CACHE_FNV64`).
|
||||
pub fn fnv1a64(&self) -> u64 {
|
||||
fnv1a64_words(&self.words)
|
||||
}
|
||||
}
|
||||
|
||||
/// Derive `ts.len()` items into `out`, all chains interleaved round by round so the cache-line misses of
|
||||
/// independent items overlap in the memory system (`deriveItems` in the Swift).
|
||||
pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) {
|
||||
let n = ts.len();
|
||||
debug_assert!(out.len() >= n);
|
||||
for k in 0..n {
|
||||
let s = &mut out[k];
|
||||
let t = ts[k];
|
||||
s[..8].copy_from_slice(&mp.key);
|
||||
for i in 0..8 {
|
||||
s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]);
|
||||
}
|
||||
}
|
||||
for r in 0..ITEM_ROUNDS {
|
||||
let rk = round_key(r);
|
||||
for s in out[..n].iter_mut() {
|
||||
mixer(s, rk, mp);
|
||||
}
|
||||
for s in out[..n].iter_mut() {
|
||||
let line = cache.line(s[0]);
|
||||
for i in 0..16 {
|
||||
s[i] ^= line[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
let rk = round_key(ITEM_ROUNDS);
|
||||
for s in out[..n].iter_mut() {
|
||||
mixer(s, rk, mp);
|
||||
}
|
||||
}
|
||||
|
||||
/// One dataset item, 16 words.
|
||||
pub fn derive_item(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] {
|
||||
let mut out = [[0u32; 16]; 1];
|
||||
derive_items(&[t], mp, cache, &mut out);
|
||||
out[0]
|
||||
}
|
||||
|
||||
/// The CPU verifier's view of the memory-hard dataset: the mixer parameters and the 256 MiB cache.
|
||||
pub struct MemhardCpu {
|
||||
pub params: MixParams,
|
||||
pub cache: Cache,
|
||||
}
|
||||
|
||||
/// Largest batch `MemhardCpu::fetch` accepts (two warps).
|
||||
pub const FETCH_MAX: usize = 64;
|
||||
|
||||
impl MemhardCpu {
|
||||
pub fn new(key: [u32; 8]) -> Self {
|
||||
Self { params: MixParams::new(key), cache: Cache::fill(key) }
|
||||
}
|
||||
pub fn for_day(day: &str) -> Self {
|
||||
Self::new(day_key(day))
|
||||
}
|
||||
/// `dataset[w] = item(w >> 4)[w & 15]`.
|
||||
pub fn word(&self, w: u32) -> u32 {
|
||||
derive_item(w >> 4, &self.params, &self.cache)[(w & 15) as usize]
|
||||
}
|
||||
/// `out[k] = dataset[idx[k]]` for every k, `idx.len() <= FETCH_MAX`. Equal items are derived once.
|
||||
/// Returns the number of distinct items derived.
|
||||
pub fn fetch(&self, idx: &[u32], out: &mut [u32]) -> usize {
|
||||
let n = idx.len();
|
||||
assert!(n <= FETCH_MAX && out.len() >= n);
|
||||
let mut uniq = [0u32; FETCH_MAX];
|
||||
let mut slot = [0u8; FETCH_MAX];
|
||||
let mut u = 0usize;
|
||||
for k in 0..n {
|
||||
let t = idx[k] >> 4;
|
||||
let found = uniq[..u].iter().position(|&x| x == t);
|
||||
let j = match found {
|
||||
Some(j) => j,
|
||||
None => {
|
||||
uniq[u] = t;
|
||||
u += 1;
|
||||
u - 1
|
||||
}
|
||||
};
|
||||
slot[k] = j as u8;
|
||||
}
|
||||
let mut items = [[0u32; 16]; FETCH_MAX];
|
||||
derive_items(&uniq[..u], &self.params, &self.cache, &mut items);
|
||||
for k in 0..n {
|
||||
out[k] = items[slot[k] as usize][(idx[k] & 15) as usize];
|
||||
}
|
||||
u
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn mix_params_for_day() {
|
||||
// MEMHARD.md section 1.4 and the igneum-genesis-mh pack.
|
||||
let mp = MixParams::for_day("2026-10-03");
|
||||
assert_eq!(mp.rot, [20, 20, 19, 4, 26, 3, 3, 27]);
|
||||
assert_eq!(mp.mul[0], 0x42146205);
|
||||
assert_eq!(mp.mul[15], 0x99cfb423);
|
||||
assert_eq!(mp.rc[0], 0xbab68293);
|
||||
assert_eq!(mp.rc[15], 0x31b49ee2);
|
||||
assert!(mp.mul.iter().all(|m| m & 1 == 1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn chacha_block_is_a_permutation_plus_feedforward() {
|
||||
let x = [1u32; 16];
|
||||
let y = chacha_block(&x);
|
||||
assert_ne!(x, y);
|
||||
let z = chacha_block(&x);
|
||||
assert_eq!(y, z);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn first_cache_line_matches_pack() {
|
||||
// vectors.json cache_head for day 2026-10-03: segment 0, line 0, with prev = 0.
|
||||
let key = day_key("2026-10-03");
|
||||
let mut words = vec![0u32; CACHE_LINES_PER_SEGMENT * 16];
|
||||
Cache::fill_segment(&mut words, 0, &key);
|
||||
assert_eq!(
|
||||
&words[..16],
|
||||
&[
|
||||
0x355a86d2, 0x7957db1c, 0xd21772af, 0x6fc1e09b, 0xd55ce61d, 0x6e6a278b, 0xd3f543ce, 0x223d8e82,
|
||||
0x143ab337, 0x2e9f05bd, 0x2eb389bf, 0x0c6e449e, 0x5cfa4222, 0xba6560fe, 0x8e3e1aa4, 0xdbcc1d53
|
||||
]
|
||||
);
|
||||
}
|
||||
}
|
||||
114
igneum-pow/src/seed.rs
Normal file
114
igneum-pow/src/seed.rs
Normal file
|
|
@ -0,0 +1,114 @@
|
|||
//! Seed derivation and the SplitMix64 stream, exactly as `proto-metal/main.swift` does them.
|
||||
//!
|
||||
//! Today a seed is a string ("igneum-genesis", "day/2026-10-03"). On the chain the epoch seed will be the
|
||||
//! output of a class-group VDF over a certified checkpoint hash. [`seed_words_from_bytes`] is the function
|
||||
//! boundary for that: whatever bytes the chain settles on go through the same FNV-1a construction, so the
|
||||
//! generator and the day key never need to know where the bytes came from.
|
||||
|
||||
/// FNV-1a 64 over `bytes` with the standard basis. Used for the cache fingerprint in the packs.
|
||||
pub fn fnv1a64(bytes: &[u8]) -> u64 {
|
||||
fnv1a64_with_basis(0xcbf29ce484222325, bytes)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn fnv1a64_with_basis(basis: u64, bytes: &[u8]) -> u64 {
|
||||
let mut h = basis;
|
||||
for &b in bytes {
|
||||
h ^= b as u64;
|
||||
h = h.wrapping_mul(0x100000001b3);
|
||||
}
|
||||
h
|
||||
}
|
||||
|
||||
/// FNV-1a 64 over 32-bit words in little-endian byte order (the cache is hashed as raw memory).
|
||||
pub fn fnv1a64_words(words: &[u32]) -> u64 {
|
||||
let mut h: u64 = 0xcbf29ce484222325;
|
||||
for &w in words {
|
||||
for b in w.to_le_bytes() {
|
||||
h ^= b as u64;
|
||||
h = h.wrapping_mul(0x100000001b3);
|
||||
}
|
||||
}
|
||||
h
|
||||
}
|
||||
|
||||
/// The 32-byte seed (8 x u32) from arbitrary bytes: FNV-1a 64 with four salts, each finalised with the
|
||||
/// murmur-style mix `h ^= h >> 33; h *= 0xff51afd7ed558ccd; h ^= h >> 33`; low word then high word.
|
||||
pub fn seed_words_from_bytes(bytes: &[u8]) -> [u32; 8] {
|
||||
let mut words = [0u32; 8];
|
||||
for salt in 0..4u64 {
|
||||
let basis = 0xcbf29ce484222325u64 ^ salt.wrapping_mul(0x9E3779B97F4A7C15);
|
||||
let mut h = fnv1a64_with_basis(basis, bytes);
|
||||
h ^= h >> 33;
|
||||
h = h.wrapping_mul(0xff51afd7ed558ccd);
|
||||
h ^= h >> 33;
|
||||
words[2 * salt as usize] = h as u32;
|
||||
words[2 * salt as usize + 1] = (h >> 32) as u32;
|
||||
}
|
||||
words
|
||||
}
|
||||
|
||||
/// The 32-byte seed from a string (its UTF-8 bytes). `seedWords` in the Swift.
|
||||
pub fn seed_words(s: &str) -> [u32; 8] {
|
||||
seed_words_from_bytes(s.as_bytes())
|
||||
}
|
||||
|
||||
/// The day key: the 8 words of `seed_words("day/" + day)`. `K` in MEMHARD.md; `d0, d1` are `K[0], K[1]`.
|
||||
pub fn day_key(day: &str) -> [u32; 8] {
|
||||
seed_words(&format!("day/{day}"))
|
||||
}
|
||||
|
||||
/// SplitMix64, the one deterministic stream every draw in the prototype comes from.
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub struct SplitMix64 {
|
||||
pub s: u64,
|
||||
}
|
||||
|
||||
impl SplitMix64 {
|
||||
pub fn new(s: u64) -> Self {
|
||||
Self { s }
|
||||
}
|
||||
#[inline]
|
||||
pub fn next(&mut self) -> u64 {
|
||||
self.s = self.s.wrapping_add(0x9E3779B97F4A7C15);
|
||||
let mut z = self.s;
|
||||
z = (z ^ (z >> 30)).wrapping_mul(0xBF58476D1CE4E5B9);
|
||||
z = (z ^ (z >> 27)).wrapping_mul(0x94D049BB133111EB);
|
||||
z ^ (z >> 31)
|
||||
}
|
||||
/// `next() % n` as the Swift `below` does it (modulo, not rejection sampling).
|
||||
#[inline]
|
||||
pub fn below(&mut self, n: u64) -> u64 {
|
||||
self.next() % n
|
||||
}
|
||||
}
|
||||
|
||||
/// The generator's stream for a seed: `(w0 | w1 << 32) ^ ((w2 | w3 << 32) * 0x9E3779B97F4A7C15)`.
|
||||
pub fn program_rng(seed: &[u32; 8]) -> SplitMix64 {
|
||||
let lo = seed[0] as u64 | ((seed[1] as u64) << 32);
|
||||
let hi = seed[2] as u64 | ((seed[3] as u64) << 32);
|
||||
SplitMix64::new(lo ^ hi.wrapping_mul(0x9E3779B97F4A7C15))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn genesis_seed_words() {
|
||||
// From proto-cuda/packs/igneum-genesis/program.json.
|
||||
assert_eq!(
|
||||
seed_words("igneum-genesis"),
|
||||
[0x67a9a7be, 0x1a155b25, 0xfddfb732, 0x4b5af2e8, 0xc55caf33, 0xa27c13b7, 0x06628a48, 0x03852469]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn day_key_2026_10_03() {
|
||||
// MEMHARD.md section 1.1.
|
||||
assert_eq!(
|
||||
day_key("2026-10-03"),
|
||||
[0x3067619f, 0x3c269176, 0x84a03b03, 0xf8c63294, 0xff977c5b, 0xe60def3e, 0x63630141, 0xb8fbcb58]
|
||||
);
|
||||
}
|
||||
}
|
||||
349
igneum-pow/src/verify.rs
Normal file
349
igneum-pow/src/verify.rs
Normal file
|
|
@ -0,0 +1,349 @@
|
|||
//! The CPU reference interpreter for one 32-lane warp (`cpuWarpTraced` in the Swift) and the API the node
|
||||
//! calls. Dataset words come from the memory-hard cache (default) or from the closed form (old packs).
|
||||
|
||||
use crate::generator::{generate, Instr, Op, Program, ITERATIONS, LANES};
|
||||
use crate::memhard::MemhardCpu;
|
||||
use crate::seed::day_key;
|
||||
|
||||
/// Dataset element, closed form of (day words, index). The original prototype's six-operation element.
|
||||
#[inline(always)]
|
||||
pub fn dataset_elem(i: u32, d0: u32, d1: u32) -> u32 {
|
||||
let mut x = i ^ d0;
|
||||
x = x.wrapping_mul(0x9E3779B1);
|
||||
x ^= x >> 15;
|
||||
x = x.wrapping_add(d1);
|
||||
x = x.wrapping_mul(0x85EBCA77);
|
||||
x ^= x >> 13;
|
||||
x = x.wrapping_mul(0xC2B2AE3D);
|
||||
x ^= x >> 16;
|
||||
x
|
||||
}
|
||||
|
||||
/// splitmix32, used for the register init.
|
||||
#[inline(always)]
|
||||
pub fn splitmix32(v: u32) -> u32 {
|
||||
let mut x = v;
|
||||
x ^= x >> 16;
|
||||
x = x.wrapping_mul(0x7feb352d);
|
||||
x ^= x >> 15;
|
||||
x = x.wrapping_mul(0x846ca68b);
|
||||
x ^= x >> 16;
|
||||
x
|
||||
}
|
||||
|
||||
/// Which construction fills the dataset words.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum DatasetMode {
|
||||
/// The six-operation closed form (packs igneum-genesis and igneum-hourly). Not memory-hard.
|
||||
ClosedForm,
|
||||
/// The 256 MiB cache and 8 dependent reads per item (pack igneum-genesis-mh, MEMHARD.md). The default.
|
||||
MemoryHard,
|
||||
}
|
||||
|
||||
impl DatasetMode {
|
||||
pub fn name(self) -> &'static str {
|
||||
match self {
|
||||
DatasetMode::ClosedForm => "closed-form",
|
||||
DatasetMode::MemoryHard => "memory-hard",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Where the interpreter reads dataset words from.
|
||||
pub enum Dataset {
|
||||
ClosedForm { d0: u32, d1: u32 },
|
||||
MemoryHard(MemhardCpu),
|
||||
}
|
||||
|
||||
/// A dataset of `2^log2` words plus the construction that fills it.
|
||||
pub struct DatasetSource {
|
||||
pub log2_words: u32,
|
||||
pub mask: u32,
|
||||
/// The day key `K`; `d0, d1 = K[0], K[1]`.
|
||||
pub key: [u32; 8],
|
||||
pub dataset: Dataset,
|
||||
}
|
||||
|
||||
impl DatasetSource {
|
||||
/// Build the source for a day. Memory-hard mode fills the 256 MiB cache on the calling thread.
|
||||
pub fn new(day: &str, mode: DatasetMode, log2_words: u32) -> Self {
|
||||
Self::from_key(day_key(day), mode, log2_words)
|
||||
}
|
||||
|
||||
pub fn from_key(key: [u32; 8], mode: DatasetMode, log2_words: u32) -> Self {
|
||||
assert!((4..=32).contains(&log2_words), "dataset log2 must be in 4..=32");
|
||||
let mask = if log2_words == 32 { u32::MAX } else { (1u32 << log2_words) - 1 };
|
||||
let dataset = match mode {
|
||||
DatasetMode::ClosedForm => Dataset::ClosedForm { d0: key[0], d1: key[1] },
|
||||
DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::new(key)),
|
||||
};
|
||||
Self { log2_words, mask, key, dataset }
|
||||
}
|
||||
|
||||
pub fn mode(&self) -> DatasetMode {
|
||||
match self.dataset {
|
||||
Dataset::ClosedForm { .. } => DatasetMode::ClosedForm,
|
||||
Dataset::MemoryHard(_) => DatasetMode::MemoryHard,
|
||||
}
|
||||
}
|
||||
|
||||
pub fn memhard(&self) -> Option<&MemhardCpu> {
|
||||
match &self.dataset {
|
||||
Dataset::MemoryHard(m) => Some(m),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// `dataset[w & mask]`.
|
||||
pub fn word(&self, w: u32) -> u32 {
|
||||
let w = w & self.mask;
|
||||
match &self.dataset {
|
||||
Dataset::ClosedForm { d0, d1 } => dataset_elem(w, *d0, *d1),
|
||||
Dataset::MemoryHard(m) => m.word(w),
|
||||
}
|
||||
}
|
||||
|
||||
/// `out[k] = dataset[idx[k]]`; indices are already masked. Returns items derived (0 for the closed form).
|
||||
#[inline]
|
||||
fn fetch(&self, idx: &[u32; LANES], out: &mut [u32; LANES]) -> usize {
|
||||
match &self.dataset {
|
||||
Dataset::ClosedForm { d0, d1 } => {
|
||||
for k in 0..LANES {
|
||||
out[k] = dataset_elem(idx[k], *d0, *d1);
|
||||
}
|
||||
0
|
||||
}
|
||||
Dataset::MemoryHard(m) => m.fetch(idx, out),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The result of interpreting one warp.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub struct WarpResult {
|
||||
pub hashes: [u64; LANES],
|
||||
/// Distinct dataset items derived from the cache (0 in closed-form mode).
|
||||
pub items_derived: usize,
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
fn mulhi32(a: u32, b: u32) -> u32 {
|
||||
((a as u64 * b as u64) >> 32) as u32
|
||||
}
|
||||
|
||||
/// Interpret `program` for the 32 nonces `base_nonce .. base_nonce + 31` (wrapping). Registers are kept
|
||||
/// register-major (`r[reg][lane]`) so the per-lane loops vectorise; the semantics are those of the Metal
|
||||
/// and CUDA kernels instruction for instruction.
|
||||
pub fn interpret_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) -> WarpResult {
|
||||
let mask = ds.mask;
|
||||
let seed = &program.seed;
|
||||
let mut r = [[0u32; LANES]; 8];
|
||||
for lane in 0..LANES {
|
||||
let nonce = base_nonce.wrapping_add(lane as u32);
|
||||
for i in 0..8 {
|
||||
let mut x = nonce ^ seed[i];
|
||||
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
|
||||
x = splitmix32(x);
|
||||
r[i][lane] = x ^ seed[(i + 1) & 7];
|
||||
}
|
||||
}
|
||||
let mut items_derived = 0usize;
|
||||
let mut idx = [0u32; LANES];
|
||||
let mut val = [0u32; LANES];
|
||||
for _ in 0..ITERATIONS {
|
||||
let sel = r[0];
|
||||
for ins in &program.instrs {
|
||||
step(ins, &mut r, &sel, mask, ds, &mut idx, &mut val, &mut items_derived);
|
||||
}
|
||||
}
|
||||
let mut hashes = [0u64; LANES];
|
||||
for lane in 0..LANES {
|
||||
let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21);
|
||||
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
|
||||
hashes[lane] = ((hi as u64) << 32) | lo as u64;
|
||||
}
|
||||
WarpResult { hashes, items_derived }
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn step(
|
||||
ins: &Instr,
|
||||
r: &mut [[u32; LANES]; 8],
|
||||
sel: &[u32; LANES],
|
||||
mask: u32,
|
||||
ds: &DatasetSource,
|
||||
idx: &mut [u32; LANES],
|
||||
val: &mut [u32; LANES],
|
||||
items_derived: &mut usize,
|
||||
) {
|
||||
let d = ins.dst as usize;
|
||||
let a = ins.src as usize;
|
||||
match ins.op {
|
||||
Op::Add => {
|
||||
let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32);
|
||||
let src = r[a];
|
||||
for lane in 0..LANES {
|
||||
let s = (sel[lane] >> bit) & 1;
|
||||
let c = if s != 0 { imm2 } else { imm };
|
||||
r[d][lane] = r[d][lane].wrapping_add(src[lane]).wrapping_add(c);
|
||||
}
|
||||
}
|
||||
Op::Sub => {
|
||||
let src = r[a];
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] = r[d][lane].wrapping_sub(src[lane]);
|
||||
}
|
||||
}
|
||||
Op::Mul => {
|
||||
let src = r[a];
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] = r[d][lane].wrapping_mul(src[lane]);
|
||||
}
|
||||
}
|
||||
Op::MulHi => {
|
||||
let src = r[a];
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] = mulhi32(r[d][lane], src[lane]);
|
||||
}
|
||||
}
|
||||
Op::Xor => {
|
||||
let src = r[a];
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] ^= src[lane];
|
||||
}
|
||||
}
|
||||
Op::Or => {
|
||||
let src = r[a];
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] |= src[lane];
|
||||
}
|
||||
}
|
||||
Op::Rotl => {
|
||||
let n = ins.rot;
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] = r[d][lane].rotate_left(n);
|
||||
}
|
||||
}
|
||||
Op::Rotr => {
|
||||
let src = r[a];
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] = r[d][lane].rotate_right(src[lane] & 31);
|
||||
}
|
||||
}
|
||||
Op::Mad => {
|
||||
let src = r[a];
|
||||
let src2 = r[ins.src2 as usize];
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] = src[lane].wrapping_mul(src2[lane]).wrapping_add(r[d][lane]);
|
||||
}
|
||||
}
|
||||
Op::Shfl => {
|
||||
let src = r[a];
|
||||
let m = ins.mask as usize;
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] ^= src[lane ^ m];
|
||||
}
|
||||
}
|
||||
Op::Load => {
|
||||
for lane in 0..LANES {
|
||||
idx[lane] = r[a][lane] & mask;
|
||||
}
|
||||
*items_derived += ds.fetch(idx, val);
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] ^= val[lane];
|
||||
}
|
||||
}
|
||||
Op::WLoad => {
|
||||
// Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l.
|
||||
let base = (r[a][0] & mask) & !31;
|
||||
for lane in 0..LANES {
|
||||
idx[lane] = base + lane as u32;
|
||||
}
|
||||
*items_derived += ds.fetch(idx, val);
|
||||
for lane in 0..LANES {
|
||||
r[d][lane] ^= val[lane];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The 32 hashes of one warp (`cpuWarp` in the Swift).
|
||||
pub fn hash_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) -> [u64; LANES] {
|
||||
interpret_warp(program, base_nonce, ds).hashes
|
||||
}
|
||||
|
||||
/// Everything a node needs to verify blocks of one epoch on one day: the program for the epoch seed and the
|
||||
/// dataset source for the day key. Building one in memory-hard mode fills the 256 MiB cache (about 0.2 s on
|
||||
/// one core); keep it for the whole epoch and share it between threads (`&Epoch` is `Send + Sync`).
|
||||
pub struct Epoch {
|
||||
pub program: Program,
|
||||
pub dataset: DatasetSource,
|
||||
}
|
||||
|
||||
/// Default dataset size: 2^28 words = 1 GiB.
|
||||
pub const DEFAULT_DATASET_LOG2: u32 = 28;
|
||||
|
||||
impl Epoch {
|
||||
pub fn new(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32) -> Self {
|
||||
Self { program: generate(seed), dataset: DatasetSource::new(day, mode, dataset_log2) }
|
||||
}
|
||||
|
||||
/// The production shape: memory-hard, 1 GiB dataset.
|
||||
pub fn memory_hard(seed: &str, day: &str) -> Self {
|
||||
Self::new(seed, day, DatasetMode::MemoryHard, DEFAULT_DATASET_LOG2)
|
||||
}
|
||||
|
||||
/// The 32 hashes of the warp starting at `base_nonce`.
|
||||
pub fn hash_warp(&self, base_nonce: u32) -> [u64; LANES] {
|
||||
hash_warp(&self.program, base_nonce, &self.dataset)
|
||||
}
|
||||
|
||||
pub fn interpret_warp(&self, base_nonce: u32) -> WarpResult {
|
||||
interpret_warp(&self.program, base_nonce, &self.dataset)
|
||||
}
|
||||
|
||||
/// The hash of one nonce. The verification unit is a warp, so the 31 sibling nonces of the aligned
|
||||
/// 32-nonce group are computed too (the shuffles couple the lanes).
|
||||
pub fn hash(&self, nonce: u32) -> u64 {
|
||||
self.hash_warp(nonce & !31)[(nonce & 31) as usize]
|
||||
}
|
||||
|
||||
/// `hash(nonce) <= target`. The hash is 64 bits; the fork maps it into its 256-bit target space.
|
||||
pub fn verify_block(&self, nonce: u32, target: u64) -> bool {
|
||||
self.hash(nonce) <= target
|
||||
}
|
||||
}
|
||||
|
||||
/// One-shot `verify_block(seed, nonce, target)`: builds the epoch (cache fill included) and checks. For a
|
||||
/// node use [`Epoch`] and keep it; this exists for scripts and tests.
|
||||
pub fn verify_block(seed: &str, day: &str, nonce: u32, target: u64) -> bool {
|
||||
Epoch::memory_hard(seed, day).verify_block(nonce, target)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn closed_form_head_matches_pack() {
|
||||
// vectors.json dataset_head for igneum-genesis (closed form), day 2026-10-03.
|
||||
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28);
|
||||
assert_eq!(ds.word(0), 0x82174c0f);
|
||||
assert_eq!(ds.word(1), 0x577bdb9c);
|
||||
assert_eq!(ds.word(0x0fffffff), 0xf78c84a4);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn closed_form_genesis_vector_lane0() {
|
||||
let e = Epoch::new("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 28);
|
||||
let w = e.hash_warp(0);
|
||||
assert_eq!(w[0], 0x2941e93c76cb1910);
|
||||
assert_eq!(w[31], 0x453388e1be04e25f);
|
||||
assert_eq!(e.hash(0), w[0]);
|
||||
assert_eq!(e.hash(31), w[31]);
|
||||
assert!(e.verify_block(0, u64::MAX));
|
||||
assert!(e.verify_block(0, w[0]));
|
||||
assert!(!e.verify_block(0, w[0] - 1));
|
||||
}
|
||||
}
|
||||
286
igneum-pow/tests/packs.rs
Normal file
286
igneum-pow/tests/packs.rs
Normal file
|
|
@ -0,0 +1,286 @@
|
|||
//! Agreement with the Swift prototype through the checked-in packs under proto-cuda/packs/.
|
||||
//!
|
||||
//! igneum-genesis-mh: memory-hard dataset (cache FNV, head and last line, 64 sampled words, 96 hashes).
|
||||
//! igneum-genesis and igneum-hourly: closed-form dataset (head, last, 64 samples, 96 hashes each).
|
||||
//! All three: program.json instruction by instruction, and every emitted source file byte for byte.
|
||||
|
||||
use igneum_pow::emit::{
|
||||
cuda_kernel, cuda_memhard_header, export_pack, metal_memhard, metal_program, opencl_kernel, program_header,
|
||||
program_json, LoadSource,
|
||||
};
|
||||
use igneum_pow::generator::{generate, Op};
|
||||
use igneum_pow::memhard::CACHE_WORDS;
|
||||
use igneum_pow::verify::{DatasetMode, Epoch};
|
||||
use serde_json::Value;
|
||||
use std::path::PathBuf;
|
||||
use std::sync::OnceLock;
|
||||
|
||||
const DAY: &str = "2026-10-03";
|
||||
|
||||
fn packs_dir() -> PathBuf {
|
||||
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs")
|
||||
}
|
||||
|
||||
fn read(pack: &str, file: &str) -> String {
|
||||
let p = packs_dir().join(pack).join(file);
|
||||
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
|
||||
}
|
||||
|
||||
/// The Swift exporter writes the cache line mask quoted inside the "item" string, which is not valid JSON.
|
||||
/// Normalise that one defect so the file can be parsed and compared; the Rust emitter writes it bare.
|
||||
fn fix_swift_item_line(s: &str) -> String {
|
||||
s.replace("& \"0x003fffff\";", "& 0x003fffff;")
|
||||
}
|
||||
|
||||
fn json(pack: &str, file: &str) -> Value {
|
||||
let text = fix_swift_item_line(&read(pack, file));
|
||||
serde_json::from_str(&text).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"))
|
||||
}
|
||||
|
||||
fn hex32(v: &Value) -> u32 {
|
||||
u32::from_str_radix(v.as_str().unwrap().trim_start_matches("0x"), 16).unwrap()
|
||||
}
|
||||
fn hex64(v: &Value) -> u64 {
|
||||
u64::from_str_radix(v.as_str().unwrap().trim_start_matches("0x"), 16).unwrap()
|
||||
}
|
||||
|
||||
/// One memory-hard epoch shared by every test (the cache fill is 256 MiB and about 0.2 s).
|
||||
fn mh_epoch() -> &'static Epoch {
|
||||
static E: OnceLock<Epoch> = OnceLock::new();
|
||||
E.get_or_init(|| Epoch::new("igneum-genesis", DAY, DatasetMode::MemoryHard, 28))
|
||||
}
|
||||
|
||||
fn closed_epoch(seed: &str) -> Epoch {
|
||||
Epoch::new(seed, DAY, DatasetMode::ClosedForm, 28)
|
||||
}
|
||||
|
||||
fn check_program_json(pack: &str) {
|
||||
let j = json(pack, "program.json");
|
||||
let seed = j["seed"].as_str().unwrap();
|
||||
let p = generate(seed);
|
||||
let sw: Vec<u32> = j["seed_words"].as_array().unwrap().iter().map(hex32).collect();
|
||||
assert_eq!(p.seed.to_vec(), sw, "{pack}: seed words");
|
||||
assert_eq!(p.loads_per_hash() as u64, j["loads_per_hash"].as_u64().unwrap(), "{pack}: loads per hash");
|
||||
let instrs = j["instructions"].as_array().unwrap();
|
||||
assert_eq!(instrs.len(), p.instrs.len(), "{pack}: instruction count");
|
||||
for (k, (ins, ji)) in p.instrs.iter().zip(instrs).enumerate() {
|
||||
assert_eq!(ji["i"].as_u64().unwrap() as usize, k);
|
||||
assert_eq!(Op::from_name(ji["op"].as_str().unwrap()).unwrap(), ins.op, "{pack} #{k} op");
|
||||
assert_eq!(ji["dst"].as_u64().unwrap(), ins.dst as u64, "{pack} #{k} dst");
|
||||
assert_eq!(ji["src"].as_u64().unwrap(), ins.src as u64, "{pack} #{k} src");
|
||||
assert_eq!(ji["src2"].as_u64().unwrap(), ins.src2 as u64, "{pack} #{k} src2");
|
||||
assert_eq!(hex32(&ji["imm"]), ins.imm, "{pack} #{k} imm");
|
||||
assert_eq!(hex32(&ji["imm2"]), ins.imm2, "{pack} #{k} imm2");
|
||||
assert_eq!(ji["rot"].as_u64().unwrap(), ins.rot as u64, "{pack} #{k} rot");
|
||||
assert_eq!(ji["bit"].as_u64().unwrap(), ins.bit as u64, "{pack} #{k} bit");
|
||||
assert_eq!(ji["mask"].as_u64().unwrap(), ins.mask as u64, "{pack} #{k} mask");
|
||||
}
|
||||
// Op mix.
|
||||
let mix = j["op_mix"].as_object().unwrap();
|
||||
for (name, count) in p.histogram() {
|
||||
assert_eq!(mix[name].as_u64().unwrap() as usize, count, "{pack}: op_mix {name}");
|
||||
}
|
||||
assert_eq!(mix.len(), p.histogram().len());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn program_json_matches_all_packs() {
|
||||
for pack in ["igneum-genesis", "igneum-genesis-mh", "igneum-hourly"] {
|
||||
check_program_json(pack);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mixer_params_match_pack() {
|
||||
let j = json("igneum-genesis-mh", "program.json");
|
||||
let mp = &mh_epoch().dataset.memhard().unwrap().params;
|
||||
let key: Vec<u32> = j["dataset"]["key"].as_array().unwrap().iter().map(hex32).collect();
|
||||
assert_eq!(mp.key.to_vec(), key);
|
||||
let rot: Vec<u32> =
|
||||
j["dataset"]["mixer"]["rot"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u32).collect();
|
||||
assert_eq!(mp.rot.to_vec(), rot);
|
||||
let mul: Vec<u32> = j["dataset"]["mixer"]["mul"].as_array().unwrap().iter().map(hex32).collect();
|
||||
assert_eq!(mp.mul.to_vec(), mul);
|
||||
let rc: Vec<u32> = j["dataset"]["mixer"]["rc"].as_array().unwrap().iter().map(hex32).collect();
|
||||
assert_eq!(mp.rc.to_vec(), rc);
|
||||
assert_eq!(hex32(&j["dataset"]["d0"]), mp.key[0]);
|
||||
assert_eq!(hex32(&j["dataset"]["d1"]), mp.key[1]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cache_matches_vectors() {
|
||||
let v = json("igneum-genesis-mh", "vectors.json");
|
||||
let m = mh_epoch().dataset.memhard().unwrap();
|
||||
let w = m.cache.words();
|
||||
assert_eq!(w.len(), CACHE_WORDS);
|
||||
let head: Vec<u32> = v["cache_head"].as_array().unwrap().iter().map(hex32).collect();
|
||||
assert_eq!(&w[..16], &head[..]);
|
||||
let last: Vec<u32> = v["cache_last_line"].as_array().unwrap().iter().map(hex32).collect();
|
||||
assert_eq!(&w[CACHE_WORDS - 16..], &last[..]);
|
||||
assert_eq!(m.cache.fnv1a64(), 0x48c4f5bf24166b2e, "cache FNV-1a 64 (MEMHARD.md)");
|
||||
assert_eq!(m.cache.fnv1a64(), hex64(&v["cache_fnv1a64"]));
|
||||
}
|
||||
|
||||
fn check_dataset_words(pack: &str, e: &Epoch) {
|
||||
let v = json(pack, "vectors.json");
|
||||
let ds = &e.dataset;
|
||||
let head: Vec<u32> = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect();
|
||||
for (i, h) in head.iter().enumerate() {
|
||||
assert_eq!(ds.word(i as u32), *h, "{pack}: dataset[{i}]");
|
||||
}
|
||||
let last_index = v["dataset_last_index"].as_u64().unwrap() as u32;
|
||||
assert_eq!(last_index, ds.mask);
|
||||
assert_eq!(ds.word(last_index), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]");
|
||||
let samples = v["dataset_samples"].as_array().unwrap();
|
||||
assert_eq!(samples.len(), 64);
|
||||
let mut n = 0;
|
||||
for s in samples {
|
||||
let idx = s["index"].as_u64().unwrap() as u32;
|
||||
assert_eq!(ds.word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]");
|
||||
n += 1;
|
||||
}
|
||||
assert_eq!(n, 64);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dataset_words_match_memhard_pack() {
|
||||
check_dataset_words("igneum-genesis-mh", mh_epoch());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dataset_words_match_closed_packs() {
|
||||
check_dataset_words("igneum-genesis", &closed_epoch("igneum-genesis"));
|
||||
check_dataset_words("igneum-hourly", &closed_epoch("igneum-hourly"));
|
||||
}
|
||||
|
||||
/// Returns the number of hashes compared (3 warps x 32 lanes = 96).
|
||||
fn check_vectors(pack: &str, e: &Epoch) -> usize {
|
||||
let v = json(pack, "vectors.json");
|
||||
assert_eq!(v["dataset_mode"].as_str().unwrap(), e.dataset.mode().name());
|
||||
assert_eq!(v["dataset_log2_words"].as_u64().unwrap() as u32, e.dataset.log2_words);
|
||||
let warps = v["warps"].as_array().unwrap();
|
||||
assert_eq!(warps.len(), 3);
|
||||
let mut n = 0;
|
||||
for w in warps {
|
||||
let base = w["base_nonce"].as_u64().unwrap() as u32;
|
||||
let expected: Vec<u64> = w["expected"].as_array().unwrap().iter().map(hex64).collect();
|
||||
let got = e.hash_warp(base);
|
||||
for lane in 0..32 {
|
||||
assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}");
|
||||
n += 1;
|
||||
}
|
||||
// The single-nonce API agrees with the warp.
|
||||
assert_eq!(e.hash(base + 7), expected[7]);
|
||||
assert!(e.verify_block(base + 7, expected[7]));
|
||||
assert!(!e.verify_block(base + 7, expected[7] - 1));
|
||||
}
|
||||
n
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vectors_memhard_96() {
|
||||
assert_eq!(check_vectors("igneum-genesis-mh", mh_epoch()), 96);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vectors_closed_form_genesis_96() {
|
||||
assert_eq!(check_vectors("igneum-genesis", &closed_epoch("igneum-genesis")), 96);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn vectors_closed_form_hourly_96() {
|
||||
assert_eq!(check_vectors("igneum-hourly", &closed_epoch("igneum-hourly")), 96);
|
||||
}
|
||||
|
||||
fn assert_same_text(pack: &str, file: &str, got: &str) {
|
||||
let want = read(pack, file);
|
||||
if got != want {
|
||||
// Find the first differing line for a readable failure.
|
||||
let (gl, wl): (Vec<&str>, Vec<&str>) = (got.lines().collect(), want.lines().collect());
|
||||
for i in 0..gl.len().max(wl.len()) {
|
||||
let g = gl.get(i).copied().unwrap_or("<eof>");
|
||||
let w = wl.get(i).copied().unwrap_or("<eof>");
|
||||
if g != w {
|
||||
panic!("{pack}/{file} differs at line {}:\n pack: {w}\n rust: {g}", i + 1);
|
||||
}
|
||||
}
|
||||
panic!("{pack}/{file} differs only in trailing bytes (len {} vs {})", got.len(), want.len());
|
||||
}
|
||||
}
|
||||
|
||||
fn check_sources(pack: &str, e: &Epoch) {
|
||||
let p = &e.program;
|
||||
let mp = e.dataset.memhard().map(|m| &m.params);
|
||||
assert_same_text(pack, "kernel.cu", &cuda_kernel(p, mp));
|
||||
assert_same_text(pack, "program.metal", &metal_program(p, e.dataset.log2_words, LoadSource::Stored));
|
||||
assert_same_text(pack, "kernel.cl", &opencl_kernel(p, mp));
|
||||
assert_same_text(pack, "program.h", &program_header(p, DAY, &e.dataset.key, e.dataset.log2_words, mp));
|
||||
if let Some(mp) = mp {
|
||||
assert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp));
|
||||
assert_same_text(pack, "memhard.metal", &metal_memhard(mp));
|
||||
}
|
||||
// program.json: byte-identical after normalising the Swift quoting defect in the "item" line.
|
||||
let want = fix_swift_item_line(&read(pack, "program.json"));
|
||||
let got = program_json(p, DAY, &e.dataset.key, e.dataset.log2_words, mp);
|
||||
assert_eq!(got, want, "{pack}/program.json");
|
||||
let _: Value = serde_json::from_str(&got).expect("Rust program.json is valid JSON");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn emitted_sources_match_memhard_pack() {
|
||||
check_sources("igneum-genesis-mh", mh_epoch());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn emitted_sources_match_closed_packs() {
|
||||
check_sources("igneum-genesis", &closed_epoch("igneum-genesis"));
|
||||
check_sources("igneum-hourly", &closed_epoch("igneum-hourly"));
|
||||
}
|
||||
|
||||
/// The whole pack as `export` writes it: vectors.json and vectors.h match the Swift ones apart from the
|
||||
/// provenance string (the Swift adds "Metal GPU cross-check PASS", which the Rust side cannot claim).
|
||||
fn check_export(pack: &str, e: &Epoch) {
|
||||
let swift_v = json(pack, "vectors.json");
|
||||
let source = swift_v["source"].as_str().unwrap();
|
||||
let out = export_pack(e, DAY, source);
|
||||
let file = |name: &str| -> &str { &out.files.iter().find(|(n, _)| n == name).unwrap().1 };
|
||||
assert_same_text(pack, "vectors.json", file("vectors.json"));
|
||||
assert_same_text(pack, "vectors.h", file("vectors.h"));
|
||||
let expected: Vec<&str> = if e.dataset.mode() == DatasetMode::MemoryHard {
|
||||
vec![
|
||||
"program.json",
|
||||
"vectors.json",
|
||||
"kernel.cu",
|
||||
"kernel.cl",
|
||||
"program.h",
|
||||
"vectors.h",
|
||||
"program.metal",
|
||||
"memhard.h",
|
||||
"memhard.metal",
|
||||
]
|
||||
} else {
|
||||
vec!["program.json", "vectors.json", "kernel.cu", "kernel.cl", "program.h", "vectors.h", "program.metal"]
|
||||
};
|
||||
assert_eq!(out.files.iter().map(|(n, _)| n.as_str()).collect::<Vec<_>>(), expected);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn export_pack_matches_memhard_pack() {
|
||||
check_export("igneum-genesis-mh", mh_epoch());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn export_pack_matches_closed_packs() {
|
||||
check_export("igneum-genesis", &closed_epoch("igneum-genesis"));
|
||||
check_export("igneum-hourly", &closed_epoch("igneum-hourly"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn item_is_independent_of_dataset_size() {
|
||||
// MEMHARD.md 1.7: a smaller dataset is a prefix of items, so dataset[w] is the same at every size.
|
||||
let big = &mh_epoch().dataset;
|
||||
let m = big.memhard().unwrap();
|
||||
for w in [0u32, 1, 15, 16, 17, 0x00ff_ffff, 0x03ff_ffff] {
|
||||
assert_eq!(big.word(w), m.word(w));
|
||||
}
|
||||
}
|
||||
Loading…
Reference in a new issue