diff --git a/igneum-pow/src/blake2b.rs b/igneum-pow/src/blake2b.rs new file mode 100644 index 000000000..4b04cc622 --- /dev/null +++ b/igneum-pow/src/blake2b.rs @@ -0,0 +1,152 @@ +//! BLAKE2b (RFC 7693), the chain's own hash family (spec 01 section 0.6), written out here so the crate keeps its +//! rule of no dependency outside the standard library. Used by class v5's state leaves (`crate::state`): +//! `blake2b_512` for a leaf digest, `blake2b_256` for the sample order. Unkeyed, no salt, no personalisation. +//! Checked against the RFC's "abc" vector and the empty-input vector in the tests. + +const IV: [u64; 8] = [ + 0x6a09e667f3bcc908, + 0xbb67ae8584caa73b, + 0x3c6ef372fe94f82b, + 0xa54ff53a5f1d36f1, + 0x510e527fade682d1, + 0x9b05688c2b3e6c1f, + 0x1f83d9abfb41bd6b, + 0x5be0cd19137e2179, +]; + +const SIGMA: [[usize; 16]; 12] = [ + [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15], + [14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3], + [11, 8, 12, 0, 5, 2, 15, 13, 10, 14, 3, 6, 7, 1, 9, 4], + [7, 9, 3, 1, 13, 12, 11, 14, 2, 6, 5, 10, 4, 0, 15, 8], + [9, 0, 5, 7, 2, 4, 10, 15, 14, 1, 11, 12, 6, 8, 3, 13], + [2, 12, 6, 10, 0, 11, 8, 3, 4, 13, 7, 5, 15, 14, 1, 9], + [12, 5, 1, 15, 14, 13, 4, 10, 0, 7, 6, 3, 9, 2, 8, 11], + [13, 11, 7, 14, 12, 1, 3, 9, 5, 0, 15, 4, 8, 6, 2, 10], + [6, 15, 14, 9, 11, 3, 0, 8, 12, 2, 13, 7, 1, 4, 10, 5], + [10, 2, 8, 4, 7, 6, 1, 5, 15, 11, 9, 14, 3, 12, 13, 0], + [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15], + [14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3], +]; + +#[inline(always)] +fn g(v: &mut [u64; 16], a: usize, b: usize, c: usize, d: usize, x: u64, y: u64) { + v[a] = v[a].wrapping_add(v[b]).wrapping_add(x); + v[d] = (v[d] ^ v[a]).rotate_right(32); + v[c] = v[c].wrapping_add(v[d]); + v[b] = (v[b] ^ v[c]).rotate_right(24); + v[a] = v[a].wrapping_add(v[b]).wrapping_add(y); + v[d] = (v[d] ^ v[a]).rotate_right(16); + v[c] = v[c].wrapping_add(v[d]); + v[b] = (v[b] ^ v[c]).rotate_right(63); +} + +fn compress(h: &mut [u64; 8], block: &[u8; 128], t: u128, last: bool) { + let mut m = [0u64; 16]; + for (i, w) in m.iter_mut().enumerate() { + *w = u64::from_le_bytes(block[i * 8..i * 8 + 8].try_into().unwrap()); + } + let mut v = [0u64; 16]; + v[..8].copy_from_slice(h); + v[8..].copy_from_slice(&IV); + v[12] ^= t as u64; + v[13] ^= (t >> 64) as u64; + if last { + v[14] = !v[14]; + } + for s in SIGMA.iter() { + g(&mut v, 0, 4, 8, 12, m[s[0]], m[s[1]]); + g(&mut v, 1, 5, 9, 13, m[s[2]], m[s[3]]); + g(&mut v, 2, 6, 10, 14, m[s[4]], m[s[5]]); + g(&mut v, 3, 7, 11, 15, m[s[6]], m[s[7]]); + g(&mut v, 0, 5, 10, 15, m[s[8]], m[s[9]]); + g(&mut v, 1, 6, 11, 12, m[s[10]], m[s[11]]); + g(&mut v, 2, 7, 8, 13, m[s[12]], m[s[13]]); + g(&mut v, 3, 4, 9, 14, m[s[14]], m[s[15]]); + } + for i in 0..8 { + h[i] ^= v[i] ^ v[i + 8]; + } +} + +/// Unkeyed BLAKE2b of `data` with an output of `out_len` bytes (1..=64), written into `out[..out_len]`. +pub fn blake2b(out: &mut [u8], out_len: usize, data: &[u8]) { + assert!((1..=64).contains(&out_len) && out.len() >= out_len); + let mut h = IV; + h[0] ^= 0x0101_0000 ^ out_len as u64; + let mut t: u128 = 0; + let n = data.len(); + // every full block but the last; the last block (possibly empty) is compressed with the final flag + let full = if n == 0 { 0 } else { (n - 1) / 128 }; + for i in 0..full { + let block: &[u8; 128] = data[i * 128..i * 128 + 128].try_into().unwrap(); + t += 128; + compress(&mut h, block, t, false); + } + let mut last = [0u8; 128]; + let rest = &data[full * 128..]; + last[..rest.len()].copy_from_slice(rest); + t += rest.len() as u128; + compress(&mut h, &last, t, true); + let mut bytes = [0u8; 64]; + for (i, w) in h.iter().enumerate() { + bytes[i * 8..i * 8 + 8].copy_from_slice(&w.to_le_bytes()); + } + out[..out_len].copy_from_slice(&bytes[..out_len]); +} + +/// BLAKE2b-512 of the concatenation of `parts`. +pub fn blake2b_512(parts: &[&[u8]]) -> [u8; 64] { + let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum()); + for p in parts { + data.extend_from_slice(p); + } + let mut out = [0u8; 64]; + blake2b(&mut out, 64, &data); + out +} + +/// BLAKE2b-256 of the concatenation of `parts`. +pub fn blake2b_256(parts: &[&[u8]]) -> [u8; 32] { + let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum()); + for p in parts { + data.extend_from_slice(p); + } + let mut out = [0u8; 32]; + blake2b(&mut out, 32, &data); + out +} + +#[cfg(test)] +mod tests { + use super::*; + + fn hex(b: &[u8]) -> String { + b.iter().map(|x| format!("{x:02x}")).collect() + } + + /// RFC 7693 appendix A ("abc"), the empty input, and a two-block input against the reference implementation's + /// known values (the three-block "The quick brown fox" vector of the BLAKE2 test suite). + #[test] + fn rfc_7693_vectors() { + assert_eq!( + hex(&blake2b_512(&[b"abc"])), + "ba80a53f981c4d0d6a2797b69f12f6e94c212f14685ac4b74b12bb6fdbffa2d17d87c5392aab792dc252d5de4533cc9518d38aa8dbf1925ab92386edd4009923" + ); + assert_eq!( + hex(&blake2b_512(&[b""])), + "786a02f742015903c6c6fd852552d272912f4740e15847618a86e217f71f5419d25e1031afee585313896444934eb04b903a685b1448b755d56f701afe9be2ce" + ); + assert_eq!(hex(&blake2b_256(&[b"abc"])), "bddd813c634239723171ef3fee98579b94964e3bb1cb3e427262c8c068d52319"); + assert_eq!(hex(&blake2b_256(&[b""])), "0e5751c026e543b2e8ab2eb06099daa1d1e5df47778f7787faab45cdf12fe3a8"); + // a 128-byte input is exactly one full block compressed as the last; 129 bytes takes two + let one = [0x61u8; 128]; + let two = [0x61u8; 129]; + assert_ne!(blake2b_512(&[&one]), blake2b_512(&[&two])); + assert_eq!(blake2b_512(&[&one[..64], &one[64..]]), blake2b_512(&[&one]), "parts concatenate"); + assert_eq!( + hex(&blake2b_512(&[b"The quick brown fox jumps over the lazy dog"])), + "a8add4bdddfd93e4877d2746e62817b116364a1fa7bc148d95090bc7333b3673f82401cf7aa2e4cb1ecd90296e3f14cb5413f8ed77be73045b13914cdcd6a918" + ); + } +} diff --git a/igneum-pow/src/emit.rs b/igneum-pow/src/emit.rs index 37437fe37..1ff1af2e8 100644 --- a/igneum-pow/src/emit.rs +++ b/igneum-pow/src/emit.rs @@ -164,7 +164,11 @@ fn program_class_header_lines(p: &Program) -> String { return String::new(); } let mut s = String::new(); - if p.program_class() == ProgramClass::V4 { + if p.program_class() == ProgramClass::V5 { + s.push_str("// Program class v5 (proof of stored state and of following, docs/design/class-v5-stored-state.md): generator version 5,\n"); + s.push_str("// class v4 over a dataset whose every item is keyed by the window's execution state (IGNEUM_STATE_* below, leaves.bin);\n"); + s.push_str("// a worker that runs another class refuses this pack, and a job line names the class it wants (class=v5 era=).\n"); + } else if p.program_class() == ProgramClass::V4 { s.push_str("// Program class v4 (Counter ASIC 3.0, docs/plans/counter-asic-3-node.md): generator version 4, class v3 plus the\n"); s.push_str("// latency-shadow block (IGNEUM_SHADOW_INSTRS x IGNEUM_SHADOW_REPS per iteration); a worker that runs another class\n"); s.push_str("// refuses this pack, and a job line names the class it wants (class=v4 era=).\n"); @@ -179,6 +183,24 @@ fn program_class_header_lines(p: &Program) -> String { s } +/// The state lines of program.h (class v5): the window's reference block and state root, the leaf count, the FNV of +/// `leaves.bin` and the file's name. Empty for every dataset without leaves, so no pinned pack changes. +fn state_header_lines(ds: &DatasetSource) -> String { + let Some(l) = ds.leaves() else { return String::new() }; + let mut s = String::new(); + s.push_str("// Class v5 state (docs/design/class-v5-stored-state.md): the window's reference chain block and the state root after it;\n"); + s.push_str("// leaves.bin holds IGNEUM_STATE_LEAVES leaves of 16 little-endian words, leaf(t) = leaves[t mod IGNEUM_STATE_LEAVES].\n"); + s.push_str(&format!("#define IGNEUM_STATE_BLOCK_HEX {}\n", jstr(&hex_bytes(&l.block)))); + s.push_str(&format!("#define IGNEUM_STATE_BLOCK_NUMBER {}\n", l.number)); + s.push_str(&format!("#define IGNEUM_STATE_ROOT_HEX {}\n", jstr(&hex_bytes(&l.root)))); + s.push_str(&format!("#define IGNEUM_STATE_LEAVES {}\n", l.n())); + s.push_str(&format!("#define IGNEUM_STATE_RECORDS {}\n", l.records_total)); + s.push_str(&format!("#define IGNEUM_STATE_SAMPLED {}\n", l.sampled as u8)); + s.push_str(&format!("#define IGNEUM_STATE_LEAVES_FNV64 {}\n", hex64(l.fnv1a64()))); + s.push_str("#define IGNEUM_STATE_LEAVES_FILE \"leaves.bin\"\n"); + s +} + /// The load class lines of program.h (empty for the lottery hash, so the pinned packs do not change). fn class_header_lines(p: &Program) -> String { if p.class.is_v2() { @@ -662,28 +684,35 @@ pub fn emit_memhard_core_layout(mp: &MixParams, dialect: CoreDialect, layout: La s.push_str("}\n"); } s.push_str(&format!("// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of (round program r, cache line s[0] & mask); round program {ITEM_ROUNDS}.\n")); - s.push_str(&format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n")); + s.push_str(&item_signature(shape.state, fn_, cptr, u, lptr)); for i in 0..8 { s.push_str(&format!(" s[{i}] = {};\n", hex(k[i]))); } for i in 0..8 { s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(mul[i]), hex(c[i]))); } + if shape.state { + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= leaf[i];\n")); + } for r in 0..ITEM_ROUNDS { s.push_str(&format!(" mh_round_{r}(s);\n")); s.push_str(&format!(" {{ {cptr} line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); for ({u} i = 0u; i < 16u; ++i) s[i] ^= line[i]; }}\n")); } s.push_str(&format!(" mh_round_{ITEM_ROUNDS}(s);\n")); s.push_str("}\n"); - return finish_memhard_core(s, layout, u, fn_, cptr); + return finish_memhard_core(s, layout, u, fn_, cptr, shape.state); } - s.push_str(&format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n")); + s.push_str(&item_signature(shape.state, fn_, cptr, u, lptr)); for i in 0..8 { s.push_str(&format!(" s[{i}] = {};\n", hex(k[i]))); } for i in 0..8 { s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(mul[i]), hex(c[i]))); } + if shape.state { + // class v5: the window's state leaf of item t, before the first mixer (docs/design/class-v5-stored-state.md) + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= leaf[i];\n")); + } s.push_str(&format!(" for ({u} r = 0u; r < {ITEM_ROUNDS}u; ++r) {{\n")); if m == 1 { s.push_str(" mh_mixer(s, 0x9E3779B9u * (r + 1u));\n"); @@ -702,22 +731,48 @@ pub fn emit_memhard_core_layout(mp: &MixParams, dialect: CoreDialect, layout: La )); } s.push_str("}\n"); - finish_memhard_core(s, layout, u, fn_, cptr) + finish_memhard_core(s, layout, u, fn_, cptr, shape.state) } -/// The tail of the memhard core: `mh_word` (and the era layout helpers) after `mh_item`. -fn finish_memhard_core(mut s: String, layout: Layout, u: &str, fn_: &str, cptr: &str) -> String { +/// The `mh_item` signature: under a state shape (class v5) the item takes its 16-word leaf (`leaves + 16 (t mod n)`). +fn item_signature(state: bool, fn_: &str, cptr: &str, u: &str, lptr: &str) -> String { + if state { + format!("{fn_} void mh_item({cptr} cache, {cptr} leaf, {u} t, {lptr} s) {{\n") + } else { + format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n") + } +} + +/// The tail of the memhard core: `mh_word` (and the era layout helpers) after `mh_item`. Under a state shape +/// `mh_word` takes the leaves and their count and derives item t's leaf as `leaves + 16 (t mod nLeaves)`. +fn finish_memhard_core(mut s: String, layout: Layout, u: &str, fn_: &str, cptr: &str, state: bool) -> String { + if state { + s.push_str("// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).\n"); + s.push_str(&format!("{fn_} {cptr} mh_leaf({cptr} leaves, {u} nLeaves, {u} t) {{ return leaves + ((t % nLeaves) * 16u); }}\n")); + } if layout.is_linear() { s.push_str("// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.\n"); - s.push_str(&format!( - "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }}\n" - )); + if state { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {cptr} leaves, {u} nLeaves, {u} w) {{ {u} s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, w >> 4u), w >> 4u, s); return s[w & 15u]; }}\n" + )); + } else { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }}\n" + )); + } } else { s.push_str(&layout_helpers(layout, u, fn_)); s.push_str("// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).\n"); - s.push_str(&format!( - "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }}\n" - )); + if state { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {cptr} leaves, {u} nLeaves, {u} w) {{ {u} s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, mh_t(w)), mh_t(w), s); return s[mh_j(w)]; }}\n" + )); + } else { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }}\n" + )); + } } s } @@ -776,12 +831,23 @@ pub fn metal_memhard_layout(mp: &MixParams, layout: Layout) -> String { s.push_str(" mh_cache_segment(cache, gid);\n"); s.push_str("}\n"); s.push_str("// One thread per 64-byte item (dataset words / 16 threads).\n"); - s.push_str( - "kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n", - ); - s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); - s.push_str(" uint s[16];\n"); - s.push_str(" mh_item(cache, gid, s);\n"); + if mp.shape.state { + s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) in buffer 2, their count in buffer 3.\n"); + s.push_str( + "kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n", + ); + s.push_str(" device const uint* leaves [[buffer(2)]], constant uint& nLeaves [[buffer(3)]],\n"); + s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); + s.push_str(" uint s[16];\n"); + s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, gid), gid, s);\n"); + } else { + s.push_str( + "kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n", + ); + s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); + s.push_str(" uint s[16];\n"); + s.push_str(" mh_item(cache, gid, s);\n"); + } s.push_str(&build_store(layout, CoreDialect::Metal, "dataset", "gid")); s.push_str("}\n"); s @@ -941,7 +1007,7 @@ fn generated_by(seed: &str) -> String { format!("// Generated by igneum-pow export (generator v{GENERATOR_VERSION}) for seed \"{seed}\". Do not edit by hand.\n") } -fn hex_bytes(b: &[u8]) -> String { +pub fn hex_bytes(b: &[u8]) -> String { b.iter().map(|x| format!("{x:02x}")).collect() } @@ -1050,11 +1116,20 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3 s.push_str(" uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;\n"); s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n"); s.push_str("}\n"); - s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n"); + if p.class.state { + s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.\n"); + s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {\n"); + } else { + s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n"); + } s.push_str(" uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;\n"); s.push_str(" if (t < nItems) {\n"); s.push_str(" uint32_t s[16];\n"); - s.push_str(" mh_item(cache, t, s);\n"); + if p.class.state { + s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);\n"); + } else { + s.push_str(" mh_item(cache, t, s);\n"); + } s.push_str(&build_store(layout, CoreDialect::Cuda, "ds", "t")); s.push_str(" }\n"); s.push_str("}\n"); @@ -1121,11 +1196,20 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3 s.push_str(" return cudaGetLastError();\n"); s.push_str("}\n"); s.push('\n'); - s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n"); - s.push_str(" if (nItems == 0u) return cudaErrorInvalidValue;\n"); + if p.class.state { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {\n"); + s.push_str(" if (nItems == 0u || nLeaves == 0u) return cudaErrorInvalidValue;\n"); + } else { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n"); + s.push_str(" if (nItems == 0u) return cudaErrorInvalidValue;\n"); + } s.push_str(" uint32_t block = 256u;\n"); s.push_str(" uint32_t grid = (nItems + block - 1u) / block;\n"); - s.push_str(" igneum_build<<>>(ds, cache, nItems);\n"); + if p.class.state { + s.push_str(" igneum_build<<>>(ds, cache, leaves, nLeaves, nItems);\n"); + } else { + s.push_str(" igneum_build<<>>(ds, cache, nItems);\n"); + } s.push_str(" return cudaGetLastError();\n"); s.push_str("}\n"); s.push('\n'); @@ -1465,11 +1549,20 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: s.push_str(" uint seg = (uint)get_global_id(0);\n"); s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n"); s.push_str("}\n"); - s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {\n"); + if p.class.state { + s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.\n"); + s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, __global const uint* leaves, uint nLeaves, uint nItems) {\n"); + } else { + s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {\n"); + } s.push_str(" uint t = (uint)get_global_id(0);\n"); s.push_str(" if (t < nItems) {\n"); s.push_str(" uint s[16];\n"); - s.push_str(" mh_item(cache, t, s);\n"); + if p.class.state { + s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);\n"); + } else { + s.push_str(" mh_item(cache, t, s);\n"); + } s.push_str(&build_store(layout, CoreDialect::OpenCl, "ds", "t")); s.push_str(" }\n"); s.push_str("}\n"); @@ -1598,6 +1691,7 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!("#define IGNEUM_OP_MIX {}\n", jstr(&p.op_mix()))); s.push_str(&program_class_header_lines(p)); s.push_str(&class_header_lines(p)); + s.push_str(&state_header_lines(ds)); s.push_str(&scratch_header_lines(p)); s.push_str(&era_header_lines(p)); s.push_str(&hot_header_lines(p)); @@ -1634,7 +1728,11 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str("#ifndef IGNEUM_NO_CUDA\n"); s.push_str("// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().\n"); s.push_str("cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);\n"); - s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);\n"); + if p.class.state { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems);\n"); + } else { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);\n"); + } if p.has_hot() { s.push_str("cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments);\n"); } @@ -1829,6 +1927,19 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!(" \"era_seed_bytes\": {},\n", jstr(&hex_bytes(era)))); } } + if let Some(l) = ds.leaves() { + s.push_str(" \"state\": {\n"); + s.push_str(&format!(" \"block\": {},\n", jstr(&hex_bytes(&l.block)))); + s.push_str(&format!(" \"block_number\": {},\n", l.number)); + s.push_str(&format!(" \"root\": {},\n", jstr(&hex_bytes(&l.root)))); + s.push_str(&format!(" \"leaves\": {},\n", l.n())); + s.push_str(&format!(" \"records\": {},\n", l.records_total)); + s.push_str(&format!(" \"sampled\": {},\n", l.sampled)); + s.push_str(&format!(" \"leaves_fnv1a64\": {},\n", jhex64(l.fnv1a64()))); + s.push_str(" \"leaf_derivation\": \"leaves[i] = Blake2b-512('igneum-sd1/' || root || i_le32 || record_i) as 16 little-endian words; item t XORs leaves[t mod leaves] into its 16 initial words before the first mixer\",\n"); + s.push_str(" \"file\": \"leaves.bin\"\n"); + s.push_str(" },\n"); + } if !p.class.is_v2() { let c = p.width_counts(); s.push_str(&format!(" \"load_class\": {},\n", jstr(&p.class.name()))); @@ -2112,6 +2223,8 @@ pub fn vectors_json( /// A program pack: the files `--export-pack` writes, as (name, text). pub struct Pack { pub files: Vec<(String, String)>, + /// Binary files beside the texts: `leaves.bin` of a class v5 pack (empty for every other pack). + pub binaries: Vec<(String, Vec)>, pub bases: Vec, pub outs: Vec<[u64; 32]>, pub vectors: PackVectors, @@ -2123,6 +2236,9 @@ impl Pack { for (name, text) in &self.files { std::fs::write(dir.join(name), text)?; } + for (name, bytes) in &self.binaries { + std::fs::write(dir.join(name), bytes)?; + } Ok(()) } } @@ -2176,7 +2292,11 @@ pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack { files.push(("memhard.h".to_string(), cuda_memhard_header(p, mp))); files.push(("memhard.metal".to_string(), metal_memhard_for(p, mp))); } - Pack { files, bases, outs, vectors: v } + let binaries = match ds.leaves() { + Some(l) => vec![("leaves.bin".to_string(), l.bytes())], + None => Vec::new(), + }; + Pack { files, binaries, bases, outs, vectors: v } } /// The dataset mode a pack was written in, from its program.json text (no JSON parser needed). diff --git a/igneum-pow/src/generator.rs b/igneum-pow/src/generator.rs index 8bbc72c9f..f256f9b84 100644 --- a/igneum-pow/src/generator.rs +++ b/igneum-pow/src/generator.rs @@ -226,6 +226,10 @@ pub struct LoadClass { /// Latency-shadow program work (Counter ASIC 3.0 item 8, measured 6 October 2026 and not adopted): `Some` adds a /// block of ALU instructions run `reps` times per iteration. `None` for every other class, class v3 included. pub shadow: Option, + /// Class v5, proof of stored state and of following (`docs/design/class-v5-stored-state.md`, 7 October 2026): + /// the item derivation XORs the window's state leaf into every item before the first mixer (`crate::state`, + /// `memhard::derive_items_leaves`). The program draw does not read it. `false` for every other class. + pub state: bool, } /// The parameters one era draws from its seed `E_n` (`docs/plans/era-layout.md` section 1.1, the proposed text of @@ -403,13 +407,13 @@ impl LoadClass { impl LoadClass { /// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash. pub const V2: LoadClass = - LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None }; + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None, state: false }; /// The construction decided for program class v3 on 5 October 2026 (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): /// version 2 loads (16 slots of one word, no scratch, no width roll, so the program stream is version 2's), the /// mixer applied 4 times per round, and the cache growth rule. Name "mx4". pub const MX4: LoadClass = - LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None }; + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None, state: false }; /// The era class over `base` (`docs/plans/era-layout.md`): the parameters drawn by [`era_draw`]; when `allowed` /// has more than one width the drawn width becomes the class mix (every load that width), otherwise the base @@ -535,6 +539,11 @@ impl LoadClass { LoadClass { shadow: Some(ShadowClass { instrs, reps }), ..self } } + /// The class with the state leaves of class v5 folded into every item ("mx8+sh256x27+state"). + pub fn with_state(self) -> LoadClass { + LoadClass { state: true, ..self } + } + /// Shadow instructions per hash (0 without a shadow). pub fn shadow_instrs_per_hash(&self) -> usize { self.shadow.map(|s| s.instrs_per_hash()).unwrap_or(0) @@ -566,6 +575,10 @@ impl LoadClass { /// "mx4": the v3 construction; a trailing "m" and "g" set the mixer multiplier and the growth rule on any /// load class, "w16m4g" for example). pub fn parse(s: &str) -> Option { + // "+state": the state leaves of class v5 over any class (the suffix is outermost) + if let Some(base) = s.strip_suffix("+state") { + return Some(LoadClass::parse(base)?.with_state()); + } // "+shx": the latency-shadow block over any class (Counter ASIC 3.0 item 8) if let Some((base, sh)) = s.rsplit_once("+sh") { let (instrs, reps) = sh.split_once('x')?; @@ -689,6 +702,10 @@ impl LoadClass { /// An era class is the base name with "-era" appended ("w4-era401998a5", "mx4-era..."). /// A hot class appends "hotk[a]" ("hot64k4", "scr4k32+hot64k4a"; measured and not adopted). pub fn name(&self) -> String { + if self.state { + // "+state": class v5's leaves are a suffix on any class, outermost + return format!("{}+state", LoadClass { state: false, ..*self }.name()); + } if let Some(sh) = self.shadow { // "+shx": the shadow block is a suffix on any class ("mx8+sh256x13") return format!("{}+sh{}x{}", LoadClass { shadow: None, ..*self }.name(), sh.instrs, sh.reps); @@ -782,6 +799,10 @@ pub const GENERATOR_VERSION_V3: u32 = 3; /// Generator version of a class v4 program (Counter ASIC 3.0, 6 October 2026, PROPOSED: `program_id(4, seed, attempt)`). pub const GENERATOR_VERSION_V4: u32 = 4; +/// Generator version of a class v5 program (proof of stored state and of following, 7 October 2026, PROPOSED: +/// `program_id(5, seed, attempt)`; `docs/design/class-v5-stored-state.md`). +pub const GENERATOR_VERSION_V5: u32 = 5; + /// The program class of an epoch (Counter ASIC 2.0, 5 October 2026, `docs/plans/counter-asic-2-rollout.md`): one /// height switch in the node, `program_class_v3_activation_daa`, rounded up to an epoch boundary, decides which /// class an epoch's program is drawn from. V2 is the lottery hash as adopted on 4 October 2026, byte for byte. @@ -794,6 +815,9 @@ pub enum ProgramClass { V2, V3, V4, + /// Class v5 (`docs/design/class-v5-stored-state.md`, behind `program_class_v5_activation_daa`): class v4's program + /// over a dataset whose every item is keyed by the window's execution state ([`V5_CLASS`]), generator 5. + V5, } /// The load class of program class v3, decided 5 October 2026 (Counter ASIC 2.0, `docs/plans/counter-asic-2-status.md` @@ -811,6 +835,12 @@ pub const V3_CLASS: LoadClass = LoadClass { era: None, hot: None, ..LoadClass::M /// for draw, so a v4 epoch's day cache and dataset are the v3 day's. Composed with the era exactly as V3 is. pub const V4_CLASS: LoadClass = LoadClass { shadow: Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps: V4_SHADOW_REPS }), ..V3_CLASS }; +/// The load class of program class v5 (`docs/design/class-v5-stored-state.md`, 7 October 2026): class v4 with the +/// state leaves of the window's reference block folded into every item of the dataset ("mx8+sh256x27+state"). The +/// program draw, the shadow block, the era draw and the ladder rung are class v4's, draw for draw; only the item +/// derivation and the program id change. +pub const V5_CLASS: LoadClass = LoadClass { state: true, ..V4_CLASS }; + /// The shadow block size of class v4 at every rung of the latency ladder (`docs/design/latency-ladder.md`): 256 /// instructions. The ladder moves the pass count alone. pub const V4_SHADOW_INSTRS: u16 = 256; @@ -829,10 +859,18 @@ pub fn v4_class_at(reps: u16) -> LoadClass { } } +/// Class v5 at a rung of the latency ladder: [`v4_class_at`] with the state leaves (`v5_class_at(0) == V5_CLASS`). +pub fn v5_class_at(reps: u16) -> LoadClass { + v4_class_at(reps).with_state() +} + /// The shadow passes of a class v4 load class at any rung of the ladder, the era draw set aside (`Some(27)` for /// [`V4_CLASS`] itself); `None` for every other class, a measurement class with another block size included. pub fn v4_rung_reps(class: &LoadClass) -> Option { let base = LoadClass { era: None, ..*class }; + if base.state { + return None; + } match base.shadow { Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps }) if LoadClass { shadow: None, ..base } == V3_CLASS => Some(reps), _ => None, @@ -863,7 +901,20 @@ pub fn generate_era(seed_string: &str, seed_bytes: &[u8], base: LoadClass, era_b /// path on `mx8+sh256x27` were stamped generator 3 and so carried the v3 control's program id (`program_id(3, seed, /// attempt)` is class-independent inside a generator version); a class v4 program is generator 4 wherever it is made. pub fn era_generator_of(base: &LoadClass) -> u32 { - if ProgramClass::of_load_class(base) == Some(ProgramClass::V4) { GENERATOR_VERSION_V4 } else { GENERATOR_VERSION_V3 } + match ProgramClass::of_load_class(base) { + Some(ProgramClass::V4) => GENERATOR_VERSION_V4, + Some(ProgramClass::V5) => GENERATOR_VERSION_V5, + _ if base.state => GENERATOR_VERSION_V5, + _ => GENERATOR_VERSION_V3, + } +} + +/// The shadow passes of a class v5 load class at any rung (the state flag set aside); `None` for every other class. +pub fn v5_rung_reps(class: &LoadClass) -> Option { + if !class.state { + return None; + } + v4_rung_reps(&LoadClass { state: false, ..*class }) } /// [`generate_era`] with the generator version stamped by the caller: 3 for class v3 over [`V3_CLASS`], 4 for @@ -882,6 +933,7 @@ impl ProgramClass { ProgramClass::V2 => LoadClass::V2, ProgramClass::V3 => V3_CLASS, ProgramClass::V4 => V4_CLASS, + ProgramClass::V5 => V5_CLASS, } } @@ -891,15 +943,22 @@ impl ProgramClass { ProgramClass::V2 => GENERATOR_VERSION, ProgramClass::V3 => GENERATOR_VERSION_V3, ProgramClass::V4 => GENERATOR_VERSION_V4, + ProgramClass::V5 => GENERATOR_VERSION_V5, } } + /// Whether the class's dataset is keyed by the window's execution state (class v5). + pub fn has_state(&self) -> bool { + *self == ProgramClass::V5 + } + /// The class of a generator version: 2, 3 and 4 are the three classes, anything else is no class this crate runs. pub fn from_generator(generator: u32) -> Option { match generator { GENERATOR_VERSION => Some(ProgramClass::V2), GENERATOR_VERSION_V3 => Some(ProgramClass::V3), GENERATOR_VERSION_V4 => Some(ProgramClass::V4), + GENERATOR_VERSION_V5 => Some(ProgramClass::V5), _ => None, } } @@ -910,6 +969,7 @@ impl ProgramClass { ProgramClass::V2 => "v2", ProgramClass::V3 => "v3", ProgramClass::V4 => "v4", + ProgramClass::V5 => "v5", } } @@ -918,6 +978,7 @@ impl ProgramClass { "v2" => Some(ProgramClass::V2), "v3" => Some(ProgramClass::V3), "v4" => Some(ProgramClass::V4), + "v5" => Some(ProgramClass::V5), _ => None, } } @@ -937,6 +998,8 @@ impl ProgramClass { Some(ProgramClass::V3) } else if base == V4_CLASS { Some(ProgramClass::V4) + } else if base == V5_CLASS { + Some(ProgramClass::V5) } else { None } @@ -1055,7 +1118,9 @@ impl Program { // pack of another rung is refused as a pack of another class is. Rung 0 keeps `program_id(4, seed, attempt)` // byte for byte, so every v4 id written before the ladder stands. let v4_rung_0 = self.generator == GENERATOR_VERSION_V4 && LoadClass { era: None, ..self.class } == V4_CLASS; - if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 { + // class v5 at rung 0 is `program_id(5, seed, attempt)`; above rung 0 the class-bearing id with "state/" + let v5_rung_0 = self.generator == GENERATOR_VERSION_V5 && LoadClass { era: None, ..self.class } == V5_CLASS; + if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 || v5_rung_0 { // Spec 01 section 1.4.6: a class v3 program's id is `program_id(3, seed, attempt)`, a class v4 program's // `program_id(4, seed, attempt)` (Counter ASIC 3.0); the generator version in the preimage separates // them from every version 2 program of the same seed @@ -1070,6 +1135,7 @@ impl Program { match self.generator { GENERATOR_VERSION_V3 => ProgramClass::V3, GENERATOR_VERSION_V4 => ProgramClass::V4, + GENERATOR_VERSION_V5 => ProgramClass::V5, _ => ProgramClass::V2, } } @@ -1137,6 +1203,10 @@ pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &L b.extend_from_slice(b"added"); } } + if class.state { + // class v5: the state leaves are part of the construction + b.extend_from_slice(b"state/"); + } fnv1a64(&b) } @@ -1414,6 +1484,8 @@ pub fn generate_from_seed_bytes_program_class(seed_string: &str, seed_bytes: &[u match (class, era_bytes) { (ProgramClass::V3, Some(era)) => return generate_era(seed_string, seed_bytes, V3_CLASS, era, &V3_ALLOWED), (ProgramClass::V4, Some(era)) => return generate_era_generator(seed_string, seed_bytes, V4_CLASS, era, &V3_ALLOWED, GENERATOR_VERSION_V4), + // Class v5: the same draw inside V5_CLASS (V4_CLASS plus the state flag, which the draw does not read), generator 5. + (ProgramClass::V5, Some(era)) => return generate_era_generator(seed_string, seed_bytes, V5_CLASS, era, &V3_ALLOWED, GENERATOR_VERSION_V5), _ => {} } let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, class.load_class()); @@ -1430,15 +1502,16 @@ pub fn generate_from_seed_bytes_program_class(seed_string: &str, seed_bytes: &[u /// and class v4 at rung 0, is [`generate_from_seed_bytes_program_class`] byte for byte. The base program, the 16 loads /// and the era draw do not move with the rung: only the pass count of the shadow block does. pub fn generate_from_seed_bytes_program_class_shadow(seed_string: &str, seed_bytes: &[u8], class: ProgramClass, era_bytes: Option<&[u8]>, shadow_reps: u16) -> Program { - if class != ProgramClass::V4 || shadow_reps == 0 || shadow_reps == V4_SHADOW_REPS { + if !matches!(class, ProgramClass::V4 | ProgramClass::V5) || shadow_reps == 0 || shadow_reps == V4_SHADOW_REPS { return generate_from_seed_bytes_program_class(seed_string, seed_bytes, class, era_bytes); } - let base = v4_class_at(shadow_reps); + // class v5 at a rung: class v4's rung with the state flag, generator 5 + let (base, generator) = if class == ProgramClass::V5 { (v5_class_at(shadow_reps), GENERATOR_VERSION_V5) } else { (v4_class_at(shadow_reps), GENERATOR_VERSION_V4) }; match era_bytes { - Some(era) => generate_era_generator(seed_string, seed_bytes, base, era, &V3_ALLOWED, GENERATOR_VERSION_V4), + Some(era) => generate_era_generator(seed_string, seed_bytes, base, era, &V3_ALLOWED, generator), None => { let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, base); - p.generator = GENERATOR_VERSION_V4; + p.generator = generator; p.era_bytes = None; p } @@ -1771,17 +1844,18 @@ mod tests { assert_eq!(v3.program_id(), program_id(GENERATOR_VERSION_V3, &v3.seed, v3.attempt)); assert_ne!(v3.program_id(), program_id(GENERATOR_VERSION, &v3.seed, v3.attempt)); assert_ne!(v3.program_id(), v2.program_id()); - for c in [ProgramClass::V2, ProgramClass::V3, ProgramClass::V4] { + for c in [ProgramClass::V2, ProgramClass::V3, ProgramClass::V4, ProgramClass::V5] { assert_eq!(ProgramClass::parse(c.name()), Some(c)); assert_eq!(ProgramClass::from_generator(c.generator_version()), Some(c)); assert_eq!(ProgramClass::from_u8(c.as_u8()), Some(c)); } assert_eq!(ProgramClass::from_generator(1), None); - assert_eq!(ProgramClass::from_generator(5), None); - assert_eq!(ProgramClass::parse("v5"), None); + assert_eq!(ProgramClass::from_generator(6), None); + assert_eq!(ProgramClass::parse("v6"), None); assert_eq!(ProgramClass::default(), ProgramClass::V2); assert_eq!(ProgramClass::V2.load_class(), LoadClass::V2); - assert!(!ProgramClass::V2.has_era() && ProgramClass::V3.has_era() && ProgramClass::V4.has_era()); + assert!(!ProgramClass::V2.has_era() && ProgramClass::V3.has_era() && ProgramClass::V4.has_era() && ProgramClass::V5.has_era()); + assert!(!ProgramClass::V4.has_state() && ProgramClass::V5.has_state()); } /// Counter ASIC 3.0 (6 October 2026): class v4 is class v3 with the latency-shadow block `sh256x27`, drawn after diff --git a/igneum-pow/src/lib.rs b/igneum-pow/src/lib.rs index 244c4ab17..cf0fd0eab 100644 --- a/igneum-pow/src/lib.rs +++ b/igneum-pow/src/lib.rs @@ -24,17 +24,20 @@ pub mod accept; pub mod bind; +pub mod blake2b; pub mod derive; pub mod emit; pub mod generator; pub mod memhard; pub mod packcheck; pub mod seed; +pub mod state; pub mod verify; pub use bind::{block_init_words, day_bytes, pow256_from_lane, target64_from_le256}; pub use accept::{check as accept_program, AcceptReport, Reject}; -pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, generate_from_seed_bytes_program_class_shadow, v4_class_at, v4_counted_ops, v4_rung_reps, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, V3_CLASS, V4_CLASS, V4_SHADOW_INSTRS, V4_SHADOW_REPS}; +pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, generate_from_seed_bytes_program_class_shadow, v4_class_at, v4_counted_ops, v4_rung_reps, v5_class_at, v5_rung_reps, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, GENERATOR_VERSION_V5, V3_CLASS, V4_CLASS, V4_SHADOW_INSTRS, V4_SHADOW_REPS, V5_CLASS}; pub use memhard::{cache_log2_words, dataset_log2_words, days_since_genesis, growth_doublings, Cache, MemhardCpu, MixParams, Shape}; pub use seed::{fnv1a64, seed_words, SplitMix64}; +pub use state::{StateLeaves, StateStream}; pub use verify::{hash_warp, interpret_warp_init, verify_block, DatasetMode, DatasetSource, Epoch}; diff --git a/igneum-pow/src/main.rs b/igneum-pow/src/main.rs index 29bb77d80..95795cd5d 100644 --- a/igneum-pow/src/main.rs +++ b/igneum-pow/src/main.rs @@ -39,6 +39,9 @@ struct Args { prehash: String, epoch_hex: Option, day_hex: Option, + /// Class v5: the window's state stream file (`--state `, the IGSD1 format of `igneum_pow::state`), whose + /// leaves every item of the dataset is keyed by. + state: Option, class: LoadClass, /// Days since genesis for the cache growth rule of a class with `growth` (0: the genesis cache). days: u64, @@ -101,7 +104,8 @@ fn usage() -> ! { \x20 --class C load class: v2 (default), mx4, mx8 (class v3: mixer x8, cache growth), dr (Counter ASIC 3.0 item 2: the per-day derivation program, dr736 = the x8-equivalent), w4, w16, w64, w64x4, p4,p16,p64[xN], m[g]\n\ \x20 also: w4, w16, w64, w64x4, p4,p16,p64[xN], m[g], +shx (latency-shadow block of S ALU instructions x R passes per iteration, Counter ASIC 3.0 item 8)\n\ \x20 --days N days since genesis for the cache growth rule of a class with it (default 0: the 2^26-word cache)\n\ - \x20 --program-class v2|v3|v4 the program class of the seam (v3 = generator 3 on V3_CLASS, v4 = generator 4 on V4_CLASS = mx8+sh256x27, the chain's own derivation; --era-hex records the era seed)\n\ + \x20 --program-class v2|v3|v4|v5 the program class of the seam (v3 = generator 3 on V3_CLASS, v4 = generator 4 on V4_CLASS = mx8+sh256x27, v5 = generator 5 on V5_CLASS = mx8+sh256x27+state, the chain's own derivation; --era-hex records the era seed)\n\ + \x20 --state class v5 (or any --class ...+state): the window's state stream (IGSD1 file, igneum-day-stream --out), whose leaves key every item\n\ \x20 --shadow-reps N class v4 at a rung of the latency ladder: the shadow block's pass count (0 = the class's own 27; docs/design/latency-ladder.md), with --program-class v4\n\ \x20 --era E era layout over --class: igneum-era-test/ or :<64 hex> (the 32-byte era seed E_n)\n\ \x20 --era-widths 4[,16,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it)" @@ -116,6 +120,7 @@ fn parse() -> Args { day: "2026-10-03".into(), out: None, closed_form: false, + state: None, dataset_log2: DEFAULT_DATASET_LOG2, warps: 20, nonce: 0, @@ -150,6 +155,7 @@ fn parse() -> Args { "--class" => a.class = LoadClass::parse(&val()).unwrap_or_else(|| usage()), "--days" => a.days = val().parse().unwrap_or_else(|_| usage()), "--program-class" => a.program_class = Some(ProgramClass::parse(&val()).unwrap_or_else(|| usage())), + "--state" => a.state = Some(val()), "--era-hex" => a.era_hex = Some(val()), "--shadow-reps" => a.shadow_reps = val().parse().unwrap_or_else(|_| usage()), "--era" => a.era = Some(parse_era(&val()).unwrap_or_else(|| usage())), @@ -216,6 +222,33 @@ fn main() { fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) { let (mut e, label) = epoch_of_class(a, mode); stamp_era(&mut e, a); + // class v5: the leaves of --state, built for the dataset's size; a state class without --state is refused here + // rather than at the first derivation + if e.program.class.state { + let Some(path) = &a.state else { + eprintln!("class {} keys every item by the window's state: give --state (igneum-day-stream --out)", e.program.class.name()); + std::process::exit(2); + }; + let stream = igneum_pow::state::StateStream::read_file(std::path::Path::new(path)).unwrap_or_else(|err| { + eprintln!("{err}"); + std::process::exit(2) + }); + let leaves = igneum_pow::state::StateLeaves::from_stream(&stream, e.dataset.log2_words); + eprintln!( + "state stream {}: chain block {} {}, root {}, {} records, {} leaves{}", + path, + stream.number, + igneum_pow::emit::hex_bytes(&stream.block), + igneum_pow::emit::hex_bytes(&stream.root), + stream.records.len(), + leaves.n(), + if leaves.sampled { " (sampled)" } else { "" } + ); + e.dataset = e.dataset.with_leaves(std::sync::Arc::new(leaves)); + } else if a.state.is_some() { + eprintln!("--state given for a class without state leaves ({}); use --program-class v5 or --class +state", e.program.class.name()); + std::process::exit(2); + } (e, label) } diff --git a/igneum-pow/src/memhard.rs b/igneum-pow/src/memhard.rs index ad42470b3..56167e2e9 100644 --- a/igneum-pow/src/memhard.rs +++ b/igneum-pow/src/memhard.rs @@ -12,6 +12,8 @@ use crate::derive::{run_round, DeriveProgram, SoaState, DERIVE_REGS, SOA_LANES}; use crate::generator::LoadClass; use crate::seed::{day_key, fnv1a64_words, SplitMix64}; +use crate::state::StateLeaves; +use std::sync::Arc; pub const CACHE_LOG2_WORDS: usize = 26; pub const CACHE_SEGMENT_LOG2_LINES: usize = 6; @@ -50,11 +52,14 @@ pub struct Shape { /// program, which replaces the `mixer_mult` applications of `M_r` in every mixer slot when non-zero. 0 for /// version 2 and class v3 (the fixed mixer). pub derive_len: u32, + /// Class v5 (`docs/design/class-v5-stored-state.md`, 7 October 2026): the item derivation XORs the window's state + /// leaf `leaf(t)` into the 16 initial words before the first mixer (`crate::state`). `false` for every other class. + pub state: bool, } impl Shape { /// Version 2: one mixer application per round, a 2^26-word cache. - pub const V2: Shape = Shape { mixer_mult: 1, cache_log2_words: CACHE_LOG2_WORDS as u32, derive_len: 0 }; + pub const V2: Shape = Shape { mixer_mult: 1, cache_log2_words: CACHE_LOG2_WORDS as u32, derive_len: 0, state: false }; /// The shape of a load class on day 0 of the chain (and on every day for a class without the growth rule). pub fn for_class(class: &LoadClass) -> Shape { @@ -68,6 +73,7 @@ impl Shape { mixer_mult: class.mixer_mult(), cache_log2_words: if class.growth { cache_log2_words(days_since_genesis) } else { CACHE_LOG2_WORDS as u32 }, derive_len: class.derive_len as u32, + state: class.state, } } @@ -374,7 +380,7 @@ impl Cache { /// are the smaller cache's segments word for word. pub fn fill_log2(key: [u32; 8], log2_words: u32) -> Cache { assert!((10..=30).contains(&log2_words), "cache log2 words must be in 10..=30"); - let shape = Shape { mixer_mult: 1, cache_log2_words: log2_words, derive_len: 0 }; + let shape = Shape { mixer_mult: 1, cache_log2_words: log2_words, derive_len: 0, state: false }; let mut words = vec![0u32; shape.cache_words()]; for seg in 0..shape.cache_segments() { Self::fill_segment(&mut words, seg, &key); @@ -505,8 +511,21 @@ impl HotTable { /// (`mp.shape.mixer_mult`) round `r` applies `M` with keys `round_key(r m + j)` for `j = 0 .. m - 1` before its /// one cache read; the final mixer applies `M` with keys `round_key(8 m + j)`. `m = 1` is version 2. pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) { + derive_items_leaves(ts, mp, cache, None, out) +} + +/// [`derive_items`] with the state leaves of class v5 (`docs/design/class-v5-stored-state.md` section 2): under a +/// shape with `state`, `leaf(t)` is XORed into the 16 initial words of item `t` before the first mixer, and the leaves +/// are required; under any other shape they must be absent. A mismatch is a programming error and panics: a dataset +/// built without the state it needs would be wrong on every item, which is the class's point. +pub fn derive_items_leaves(ts: &[u32], mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) { + match (mp.shape.state, leaves) { + (true, None) => panic!("class v5 item derivation needs the window's state leaves and was given none"), + (false, Some(_)) => panic!("state leaves given to an item derivation whose shape has no state"), + _ => {} + } if let Some(prog) = &mp.derive { - return derive_items_program(ts, mp, prog, cache, out); + return derive_items_program(ts, mp, prog, cache, leaves, out); } // The item loop lives in its own function, one instance per cache size the growth rule can reach with the line // mask a constant, never inlined into the callers. Inlined into `MemhardCpu::fetch` it ran at 1.33 ms per unit @@ -515,19 +534,19 @@ pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; // constant with the loop still inlined, all stayed at 1.33; the out-of-line instances read 0.60 to 0.62). Any // other cache size (tests) takes the instance with the run-time mask. match cache.log2_words { - 26 => derive_items_mask::<{ (1u32 << 22) - 1 }>(ts, mp, cache, out), - 27 => derive_items_mask::<{ (1u32 << 23) - 1 }>(ts, mp, cache, out), - 28 => derive_items_mask::<{ (1u32 << 24) - 1 }>(ts, mp, cache, out), - 29 => derive_items_mask::<{ (1u32 << 25) - 1 }>(ts, mp, cache, out), - 30 => derive_items_mask::<{ (1u32 << 26) - 1 }>(ts, mp, cache, out), - _ => derive_items_mask::<0>(ts, mp, cache, out), + 26 => derive_items_mask::<{ (1u32 << 22) - 1 }>(ts, mp, cache, leaves, out), + 27 => derive_items_mask::<{ (1u32 << 23) - 1 }>(ts, mp, cache, leaves, out), + 28 => derive_items_mask::<{ (1u32 << 24) - 1 }>(ts, mp, cache, leaves, out), + 29 => derive_items_mask::<{ (1u32 << 25) - 1 }>(ts, mp, cache, leaves, out), + 30 => derive_items_mask::<{ (1u32 << 26) - 1 }>(ts, mp, cache, leaves, out), + _ => derive_items_mask::<0>(ts, mp, cache, leaves, out), } } /// [`derive_items`] with the cache line mask as a constant (`LINE_MASK = 0`: the cache's own run-time mask). Kept /// out of line on purpose (see [`derive_items`]). #[inline(never)] -fn derive_items_mask(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) { +fn derive_items_mask(ts: &[u32], mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) { let n = ts.len(); debug_assert!(out.len() >= n); debug_assert!(LINE_MASK == 0 || LINE_MASK == cache.line_mask); @@ -539,6 +558,13 @@ fn derive_items_mask(ts: &[u32], mp: &MixParams, cache: &C for i in 0..8 { s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); } + if let Some(l) = leaves { + // class v5: the window's state leaf of item t, before the first mixer + let leaf = l.leaf(t); + for i in 0..16 { + s[i] ^= leaf[i]; + } + } } for r in 0..ITEM_ROUNDS { for j in 0..m { @@ -570,7 +596,7 @@ fn derive_items_mask(ts: &[u32], mp: &MixParams, cache: &C /// cache reads of the batch are issued together, as in the fixed-mixer loop, so the 8 dependent misses of /// independent items overlap in the memory system. #[inline(never)] -pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, cache: &Cache, out: &mut [[u32; 16]]) { +pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) { let n = ts.len(); debug_assert!(out.len() >= n && n <= SOA_LANES); assert_eq!(prog.rounds.len(), ITEM_ROUNDS + 1); @@ -581,6 +607,12 @@ pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, ca st[i][k] = mp.key[i]; st[8 + i][k] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); } + if let Some(l) = leaves { + let leaf = l.leaf(t); + for i in 0..16 { + st[i][k] ^= leaf[i]; + } + } } let mask = cache.line_mask(); for r in 0..ITEM_ROUNDS { @@ -602,15 +634,22 @@ pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, ca /// One dataset item, 16 words. pub fn derive_item(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] { + derive_item_leaves(t, mp, cache, None) +} + +/// [`derive_item`] with the state leaves of class v5. +pub fn derive_item_leaves(t: u32, mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>) -> [u32; 16] { let mut out = [[0u32; 16]; 1]; - derive_items(&[t], mp, cache, &mut out); + derive_items_leaves(&[t], mp, cache, leaves, &mut out); out[0] } -/// The CPU verifier's view of the memory-hard dataset: the mixer parameters (with the shape) and the cache. +/// The CPU verifier's view of the memory-hard dataset: the mixer parameters (with the shape), the cache (shared, so +/// a class v5 window refresh keeps the day's 256 MiB and swaps the leaves) and, under class v5, the window's leaves. pub struct MemhardCpu { pub params: MixParams, - pub cache: Cache, + pub cache: Arc, + pub leaves: Option>, } /// Largest batch `MemhardCpu::fetch` accepts (two warps). @@ -622,7 +661,7 @@ impl MemhardCpu { Self::with_shape(key, Shape::V2) } pub fn with_shape(key: [u32; 8], shape: Shape) -> Self { - Self { params: MixParams::with_shape(key, shape), cache: Cache::fill_log2(key, shape.cache_log2_words) } + Self { params: MixParams::with_shape(key, shape), cache: Arc::new(Cache::fill_log2(key, shape.cache_log2_words)), leaves: None } } pub fn for_day(day: &str) -> Self { Self::new(day_key(day)) @@ -630,6 +669,17 @@ impl MemhardCpu { pub fn shape(&self) -> Shape { self.params.shape } + /// This view with the window's state leaves (class v5). The shape must have `state`. + pub fn with_leaves(mut self, leaves: Arc) -> Self { + assert!(self.params.shape.state, "state leaves on a shape without state"); + self.leaves = Some(leaves); + self + } + /// A view of the same day (the same cache, shared) with other leaves: the class v5 window refresh. + pub fn refreshed(&self, leaves: Arc) -> Self { + assert!(self.params.shape.state, "state leaves on a shape without state"); + Self { params: self.params.clone(), cache: self.cache.clone(), leaves: Some(leaves) } + } /// `dataset[w] = item(w >> 4)[w & 15]` (the linear layout). pub fn word(&self, w: u32) -> u32 { self.word_at(Layout::LINEAR, w) @@ -638,7 +688,7 @@ impl MemhardCpu { /// day's, so one cache serves every era of a day). pub fn word_at(&self, layout: Layout, w: u32) -> u32 { let (t, j) = layout.split(w); - derive_item(t, &self.params, &self.cache)[j as usize] + derive_item_leaves(t, &self.params, &self.cache, self.leaves.as_deref())[j as usize] } /// `out[k] = dataset[idx[k]]` for every k, `idx.len() <= FETCH_MAX`. Equal items are derived once. /// Returns the number of distinct items derived. @@ -664,7 +714,7 @@ impl MemhardCpu { slot[k] = j as u8; } let mut items = [[0u32; 16]; FETCH_MAX]; - derive_items(&uniq[..u], &self.params, &self.cache, &mut items); + derive_items_leaves(&uniq[..u], &self.params, &self.cache, self.leaves.as_deref(), &mut items); for k in 0..n { out[k] = items[slot[k] as usize][word[k] as usize]; } @@ -695,7 +745,7 @@ impl MemhardCpu { slot[k] = j as u8; } let mut items = [[0u32; 16]; FETCH_MAX]; - derive_items(&uniq[..u], &self.params, &self.cache, &mut items); + derive_items_leaves(&uniq[..u], &self.params, &self.cache, self.leaves.as_deref(), &mut items); for k in 0..n { let o = word[k] as usize; out[k][..width].copy_from_slice(&items[slot[k] as usize][o..o + width]); @@ -851,7 +901,7 @@ mod tests { let v2 = Shape::for_class_day(&LoadClass::V2, 100_000); assert_eq!(v2, Shape::V2); let v3 = Shape::for_class_day(&LoadClass::MX4, 0); - assert_eq!(v3, Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0 }); + assert_eq!(v3, Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0, state: false }); assert_eq!(Shape::for_class_day(&LoadClass::MX4, 1_460).cache_log2_words, 27); assert_eq!(v3.mixers_per_item(), 36); assert_eq!(Shape::V2.mixers_per_item(), 9); @@ -872,7 +922,7 @@ mod tests { assert_eq!(small.segments(), 64); assert_eq!(small.line_mask(), 4095); for m in [1u32, 2, 4] { - let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 16, derive_len: 0 }); + let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 16, derive_len: 0, state: false }); for t in [0u32, 1, 12_345, u32::MAX] { let got = derive_item(t, &mp, &small); let mut s = [0u32; 16]; @@ -895,8 +945,8 @@ mod tests { assert_eq!(got, s, "m {m} t {t}"); } } - let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0 }); - let v3 = MixParams::with_shape(key, Shape { mixer_mult: 4, cache_log2_words: 16, derive_len: 0 }); + let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0, state: false }); + let v3 = MixParams::with_shape(key, Shape { mixer_mult: 4, cache_log2_words: 16, derive_len: 0, state: false }); assert_ne!(derive_item(0, &v2, &small), derive_item(0, &v3, &small)); assert_eq!(round_key_mult(0, 0, 1), round_key(0)); assert_eq!(round_key_mult(8, 0, 1), round_key(8)); diff --git a/igneum-pow/src/packcheck.rs b/igneum-pow/src/packcheck.rs index 177d74b3a..6a3d4e5a7 100644 --- a/igneum-pow/src/packcheck.rs +++ b/igneum-pow/src/packcheck.rs @@ -29,6 +29,8 @@ pub struct PackIdentity { pub class: ProgramClass, /// `IGNEUM_ERA_SEED_HEX` when the pack carries one (class v3 chain packs). pub era_hex: Option, + /// `IGNEUM_STATE_ROOT_HEX` of a class v5 pack (the window's state root the leaves derive from). + pub state_root_hex: Option, } /// Why a pack is not the one a worker should mine with. `Display` is the plain-words line the logs carry. @@ -193,8 +195,18 @@ pub fn verify_pack_texts_chain( // without the block is no v4 pack. A generator 2 pack with a shadow (the measurement ladder of // proto-cuda/packs-ca3-shadow) carries a class-bearing id and stays loadable. let shadow = define_u32(program_h, "IGNEUM_SHADOW_INSTRS").unwrap_or(0); + // Class v5 (docs/design/class-v5-stored-state.md): the state lines are the mark of class v5, so a generator 5 pack + // carries IGNEUM_STATE_ROOT_HEX (and the shadow block of v4) and no other generator does. + let state_root_hex = define_str(program_h, "IGNEUM_STATE_ROOT_HEX"); + match (class, state_root_hex.is_some()) { + (ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_STATE_ROOT_HEX: not a class v5 pack".into())), + (ProgramClass::V5, true) => {} + (_, true) => return Err(PackFault::Disagree(format!("IGNEUM_GENERATOR {generator} with class v5 state lines: a program over state leaves is generator 5 (export the pack as class v5)"))), + _ => {} + } match (class, shadow > 0) { (ProgramClass::V4, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 4 (class v4) without IGNEUM_SHADOW_INSTRS: not a class v4 pack".into())), + (ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_SHADOW_INSTRS: not a class v5 pack".into())), (ProgramClass::V3, true) => { return Err(PackFault::Disagree(format!("IGNEUM_GENERATOR 3 (class v3) with a shadow block (IGNEUM_SHADOW_INSTRS {shadow}): a class v4 program is generator 4 (export the pack as class v4)"))) } @@ -256,7 +268,7 @@ pub fn verify_pack_texts_chain( if epoch_hex != want_epoch_hex || day_hex != want_day_hex { return Err(PackFault::OutOfDate { pack_epoch: epoch_hex, pack_day: day_hex, want_epoch: want_epoch_hex, want_day: want_day_hex }); } - Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw, generator, class, era_hex }) + Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw, generator, class, era_hex, state_root_hex }) } /// [`verify_pack_texts`] over a pack directory. diff --git a/igneum-pow/src/state.rs b/igneum-pow/src/state.rs new file mode 100644 index 000000000..4298b7fde --- /dev/null +++ b/igneum-pow/src/state.rs @@ -0,0 +1,287 @@ +//! Class v5, proof of stored state and of following (`docs/design/class-v5-stored-state.md`, 7 October 2026): the +//! leaves the item derivation XORs in (section 2 of the page), built from the canonical state stream of the +//! window's reference block. +//! +//! `D[i] = Blake2b-512("igneum-sd1/" || root || i_le32 || record_i)` for the `n` records of the stream, and item +//! `t` takes `leaf(t) = D[t mod n]`: every item is keyed by the state, so a hasher without it is wrong on every +//! item (the known-failed case, the first test). When the stream has more records than the dataset has items, the +//! records are ordered by `Blake2b-256("igneum-sd1-sample/" || root || record)` and the first `items` are taken, a +//! sample nobody can choose without the whole state and the root. +//! +//! The stream file (`StateStream`): the plain format every side reads without a serialisation library, `IGSD1\0`, +//! the chain block number (le64) and hash (32), the state root (32), the record count (le32), then each record as +//! its length (le32) and bytes. The node's executor writes it (`igneum/exec/src/day_stream.rs`), the miner fetches +//! it, the CLI's `--state` reads it, and a pack carries the leaves it yields as `leaves.bin`. + +use crate::blake2b::{blake2b_256, blake2b_512}; +use crate::seed::fnv1a64_words; + +pub const LEAF_TAG: &[u8] = b"igneum-sd1/"; +pub const SAMPLE_TAG: &[u8] = b"igneum-sd1-sample/"; +pub const STREAM_MAGIC: &[u8; 6] = b"IGSD1\0"; + +/// The canonical state stream at one chain block: what the executor serialises and what the leaves derive from. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct StateStream { + pub number: u64, + pub block: [u8; 32], + pub root: [u8; 32], + pub records: Vec>, +} + +impl StateStream { + pub fn encode(&self) -> Vec { + let mut b = Vec::with_capacity(6 + 8 + 32 + 32 + 4 + self.records.iter().map(|r| 4 + r.len()).sum::()); + b.extend_from_slice(STREAM_MAGIC); + b.extend_from_slice(&self.number.to_le_bytes()); + b.extend_from_slice(&self.block); + b.extend_from_slice(&self.root); + b.extend_from_slice(&(self.records.len() as u32).to_le_bytes()); + for r in &self.records { + b.extend_from_slice(&(r.len() as u32).to_le_bytes()); + b.extend_from_slice(r); + } + b + } + + pub fn decode(bytes: &[u8]) -> Result { + if bytes.len() < 6 + 8 + 32 + 32 + 4 || &bytes[..6] != STREAM_MAGIC { + return Err("not a state stream file (magic IGSD1)".into()); + } + let mut at = 6; + let number = u64::from_le_bytes(bytes[at..at + 8].try_into().unwrap()); + at += 8; + let block: [u8; 32] = bytes[at..at + 32].try_into().unwrap(); + at += 32; + let root: [u8; 32] = bytes[at..at + 32].try_into().unwrap(); + at += 32; + let n = u32::from_le_bytes(bytes[at..at + 4].try_into().unwrap()) as usize; + at += 4; + let mut records = Vec::with_capacity(n.min(1 << 20)); + for i in 0..n { + if at + 4 > bytes.len() { + return Err(format!("state stream truncated at record {i} of {n}")); + } + let len = u32::from_le_bytes(bytes[at..at + 4].try_into().unwrap()) as usize; + at += 4; + if at + len > bytes.len() { + return Err(format!("state stream truncated inside record {i} of {n}")); + } + records.push(bytes[at..at + len].to_vec()); + at += len; + } + if at != bytes.len() { + return Err(format!("state stream has {} trailing bytes", bytes.len() - at)); + } + Ok(StateStream { number, block, root, records }) + } + + pub fn read_file(path: &std::path::Path) -> Result { + let bytes = std::fs::read(path).map_err(|e| format!("read {}: {e}", path.display()))?; + Self::decode(&bytes) + } +} + +/// `D[i]`: the 64-byte digest of record `i` under `root`, as 16 little-endian words. +pub fn leaf_digest(root: &[u8; 32], i: u32, record: &[u8]) -> [u32; 16] { + let d = blake2b_512(&[LEAF_TAG, root, &i.to_le_bytes(), record]); + let mut w = [0u32; 16]; + for (k, x) in w.iter_mut().enumerate() { + *x = u32::from_le_bytes(d[k * 4..k * 4 + 4].try_into().unwrap()); + } + w +} + +/// The sample order key of a record under `root`. +pub fn sample_key(root: &[u8; 32], record: &[u8]) -> [u8; 32] { + blake2b_256(&[SAMPLE_TAG, root, record]) +} + +/// The leaves of one window (or day) of class v5: `n` digests of 64 bytes, `leaf(t) = D[t mod n]`. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct StateLeaves { + pub root: [u8; 32], + pub block: [u8; 32], + pub number: u64, + /// Records in the stream before any sample. + pub records_total: u64, + /// Whether the stream had more records than the dataset has items (the sample rule applied). + pub sampled: bool, + leaves: Vec<[u32; 16]>, +} + +impl StateLeaves { + /// The items a dataset of `2^log2_words` words has: `2^(log2_words - 4)`. + pub fn items_of(log2_words: u32) -> u64 { + 1u64 << log2_words.saturating_sub(4) + } + + /// The leaves of `records` (canonical order) under `root` for a dataset of `2^log2_words` words. An empty stream + /// yields one leaf, the digest of the empty record, so `n` is never 0. + pub fn build(root: [u8; 32], block: [u8; 32], number: u64, records: &[Vec], log2_words: u32) -> StateLeaves { + let items = Self::items_of(log2_words); + let records_total = records.len() as u64; + let empty: Vec> = vec![Vec::new()]; + let records = if records.is_empty() { &empty[..] } else { records }; + let sampled = records.len() as u64 > items; + let chosen: Vec<&Vec> = if sampled { + let mut keyed: Vec<([u8; 32], &Vec)> = records.iter().map(|r| (sample_key(&root, r), r)).collect(); + keyed.sort_unstable_by(|a, b| a.0.cmp(&b.0).then_with(|| a.1.cmp(b.1))); + keyed.into_iter().take(items as usize).map(|(_, r)| r).collect() + } else { + records.iter().collect() + }; + let leaves = chosen.iter().enumerate().map(|(i, r)| leaf_digest(&root, i as u32, r)).collect(); + StateLeaves { root, block, number, records_total, sampled, leaves } + } + + pub fn from_stream(s: &StateStream, log2_words: u32) -> StateLeaves { + Self::build(s.root, s.block, s.number, &s.records, log2_words) + } + + /// Leaves from the raw words of a `leaves.bin` (16 words per leaf), for a worker or a test that holds no stream. + pub fn from_words(root: [u8; 32], block: [u8; 32], number: u64, words: &[u32]) -> StateLeaves { + assert!(!words.is_empty() && words.len() % 16 == 0, "leaves are 16 words each"); + let leaves = words.chunks_exact(16).map(|c| c.try_into().unwrap()).collect::>(); + StateLeaves { root, block, number, records_total: leaves.len() as u64, sampled: false, leaves } + } + + #[inline(always)] + pub fn n(&self) -> u32 { + self.leaves.len() as u32 + } + + /// `leaf(t) = D[t mod n]`. + #[inline(always)] + pub fn leaf(&self, t: u32) -> &[u32; 16] { + &self.leaves[(t % self.n()) as usize] + } + + pub fn leaves(&self) -> &[[u32; 16]] { + &self.leaves + } + + /// The flat words of `leaves.bin`. + pub fn words(&self) -> Vec { + self.leaves.iter().flat_map(|l| l.iter().copied()).collect() + } + + /// The bytes of `leaves.bin` (little-endian words). + pub fn bytes(&self) -> Vec { + self.words().iter().flat_map(|w| w.to_le_bytes()).collect() + } + + /// FNV-1a 64 over the leaves as little-endian bytes (the pack's `IGNEUM_STATE_LEAVES_FNV64`). + pub fn fnv1a64(&self) -> u64 { + fnv1a64_words(&self.words()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::generator::{generate_class, V5_CLASS}; + use crate::memhard::{derive_item_leaves, Cache, MixParams, Shape}; + use crate::seed::day_key; + use crate::verify::{hash_warp, DatasetMode, DatasetSource}; + use std::sync::Arc; + + fn records(n: usize, salt: u8) -> Vec> { + (0..n).map(|i| vec![salt, i as u8, (i >> 8) as u8, 7]).collect() + } + + fn leaves(root: u8, n: usize, log2_words: u32) -> Arc { + Arc::new(StateLeaves::build([root; 32], [0x22; 32], 5, &records(n, root), log2_words)) + } + + /// The known-failed case, first: a hasher without the state (no leaves, the leaves of another root, the leaves + /// of a stream one record short, the previous window's leaves) is wrong on every item and every lane. + #[test] + fn a_stateless_hasher_is_wrong_on_every_item() { + let key = day_key("2026-10-03"); + let shape = Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: true }; + let cache = Arc::new(Cache::fill_log2(key, 16)); + let mp = MixParams::with_shape(key, shape); + let good = leaves(0x11, 93, 20); + let other_root = leaves(0x12, 93, 20); + let one_short = Arc::new(StateLeaves::build([0x13; 32], [0x22; 32], 5, &records(92, 0x11), 20)); // a record short means another root + let previous_window = leaves(0x10, 93, 20); + for (name, bad) in [("another root", other_root.clone()), ("one record short", one_short.clone()), ("the previous window", previous_window.clone())] { + let equal = (0..64u32).filter(|&t| derive_item_leaves(t * 7919, &mp, &cache, Some(&good)) == derive_item_leaves(t * 7919, &mp, &cache, Some(&bad))).count(); + assert_eq!(equal, 0, "{name}: {equal} of 64 items equal"); + } + let stateless = Shape { state: false, ..shape }; + let mp_stateless = MixParams::with_shape(key, stateless); + let equal = (0..64u32).filter(|&t| derive_item_leaves(t * 7919, &mp, &cache, Some(&good)) == derive_item_leaves(t * 7919, &mp_stateless, &cache, None)).count(); + assert_eq!(equal, 0, "no leaves at all: {equal} of 64 items equal"); + // the warp: a class v5 program over a small dataset, the same program and cache, other leaves + let program = generate_class("igneum-genesis", V5_CLASS); + let ds = DatasetSource::new_shape("2026-10-03", DatasetMode::MemoryHard, 20, shape).with_leaves(good.clone()); + let ds_other = DatasetSource::new_shape("2026-10-03", DatasetMode::MemoryHard, 20, shape).with_leaves(previous_window.clone()); + let a = hash_warp(&program, 0, &ds); + let b = hash_warp(&program, 0, &ds_other); + assert_eq!(a.iter().zip(b.iter()).filter(|(x, y)| x == y).count(), 0, "0 of 32 lanes agree"); + assert_eq!(hash_warp(&program, 0, &ds), a, "the same leaves hash the same"); + } + + /// Every item takes a leaf: `leaf(t) = D[t mod n]`, so items `t` and `t + n` share a leaf and still differ. + #[test] + fn every_item_is_keyed_and_the_leaf_wraps() { + let l = leaves(0x11, 93, 28); + assert_eq!(l.n(), 93); + assert!(!l.sampled); + assert_eq!(l.records_total, 93); + for t in [0u32, 1, 92, 93, 94, 1_000_000, u32::MAX] { + assert_eq!(l.leaf(t), l.leaf(t % 93)); + assert_eq!(*l.leaf(t), leaf_digest(&[0x11; 32], t % 93, &records(93, 0x11)[(t % 93) as usize])); + } + let key = day_key("2026-10-03"); + let shape = Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: true }; + let cache = Cache::fill_log2(key, 16); + let mp = MixParams::with_shape(key, shape); + assert_ne!(derive_item_leaves(5, &mp, &cache, Some(&l)), derive_item_leaves(5 + 93, &mp, &cache, Some(&l))); + // an empty stream yields one leaf (the digest of the empty record), never a division by zero + let empty = StateLeaves::build([0x11; 32], [0; 32], 0, &[], 28); + assert_eq!(empty.n(), 1); + assert_eq!(empty.records_total, 0); + assert_eq!(*empty.leaf(12_345), leaf_digest(&[0x11; 32], 0, &[])); + } + + /// Above the dataset size the records are sampled in the keyed order: a different root picks a different set, + /// and the set cannot be the first `items` records of the stream. + #[test] + fn the_sample_above_the_dataset_size_is_keyed_by_the_root() { + let recs = records(40, 0x33); + let a = StateLeaves::build([0x11; 32], [0; 32], 0, &recs, 8); + let b = StateLeaves::build([0x12; 32], [0; 32], 0, &recs, 8); + assert_eq!(StateLeaves::items_of(8), 16); + assert_eq!((a.n(), a.sampled, a.records_total), (16, true, 40)); + assert_ne!(a.leaves(), b.leaves(), "another root, another sample"); + // the positional first 16 are not the sample (with overwhelming probability for 40 choose 16) + let positional = StateLeaves::build([0x11; 32], [0; 32], 0, &recs[..16], 8); + assert_ne!(a.leaves(), positional.leaves()); + // the same inputs sample the same + assert_eq!(StateLeaves::build([0x11; 32], [0; 32], 0, &recs, 8), a); + // at the dataset size exactly, no sample + let c = StateLeaves::build([0x11; 32], [0; 32], 0, &recs[..16], 8); + assert!(!c.sampled && c.n() == 16); + } + + #[test] + fn stream_file_round_trip_and_refusals() { + let s = StateStream { number: 159_357, block: [0xaf; 32], root: [0x1c; 32], records: records(93, 1) }; + let bytes = s.encode(); + assert_eq!(&bytes[..6], STREAM_MAGIC); + assert_eq!(StateStream::decode(&bytes).unwrap(), s); + assert!(StateStream::decode(&bytes[..bytes.len() - 1]).is_err(), "truncated"); + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(StateStream::decode(&trailing).is_err(), "trailing bytes"); + assert!(StateStream::decode(b"IGSD0\0").is_err(), "wrong magic"); + let l = StateLeaves::from_stream(&s, 28); + let back = StateLeaves::from_words(s.root, s.block, s.number, &l.words()); + assert_eq!(back.leaves(), l.leaves()); + assert_eq!(l.bytes().len(), 93 * 64); + assert_eq!(l.fnv1a64(), back.fnv1a64()); + } +} diff --git a/igneum-pow/src/verify.rs b/igneum-pow/src/verify.rs index 34692b684..1197d4096 100644 --- a/igneum-pow/src/verify.rs +++ b/igneum-pow/src/verify.rs @@ -227,6 +227,33 @@ impl DatasetSource { Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None } } + /// This source with the window's state leaves (class v5, `docs/design/class-v5-stored-state.md`): memory-hard mode + /// under a shape with `state` only. + pub fn with_leaves(mut self, leaves: std::sync::Arc) -> Self { + match &mut self.dataset { + Dataset::MemoryHard(m) => { + assert!(m.params.shape.state, "state leaves on a dataset whose shape has no state"); + m.leaves = Some(leaves); + } + Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"), + } + self + } + + /// A source of the same day with other leaves, the 256 MiB cache shared (the class v5 window refresh). + pub fn refreshed(&self, leaves: std::sync::Arc) -> Self { + let dataset = match &self.dataset { + Dataset::MemoryHard(m) => Dataset::MemoryHard(m.refreshed(leaves)), + Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"), + }; + Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } + } + + /// The window's state leaves, when the source carries them. + pub fn leaves(&self) -> Option<&std::sync::Arc> { + self.memhard().and_then(|m| m.leaves.as_ref()) + } + /// This source with the hot table of the epoch whose program seed bytes are `seed_bytes` (`mb` MiB). pub fn with_hot(mut self, seed_bytes: &[u8], mb: u32) -> Self { self.hot = Some(HotTable::for_seed_bytes(seed_bytes, mb)); diff --git a/igneum-pow/tests/derive.rs b/igneum-pow/tests/derive.rs index 66b58df74..a64af0292 100644 --- a/igneum-pow/tests/derive.rs +++ b/igneum-pow/tests/derive.rs @@ -43,7 +43,7 @@ fn item_by_hand(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] { fn derived_item_by_hand_and_in_batches() { let key = day_key(DAY); let cache = Cache::fill_log2(key, 16); - let shape = Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8 }; + let shape = Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false }; let mp = MixParams::with_shape(key, shape); let prog = mp.derive.as_ref().unwrap(); assert_eq!(prog.rounds.len(), DERIVE_PROGRAMS); @@ -64,7 +64,7 @@ fn derived_item_by_hand_and_in_batches() { derive_items(&ts[..5], &mp, &cache, &mut out5); assert_eq!(&out5[..], &out[..5]); // the fixed mixer of the same key gives other items - let v3 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0 }); + let v3 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: false }); assert!(v3.derive.is_none()); assert_ne!(derive_item(0, &v3, &cache), derive_item(0, &mp, &cache)); } @@ -79,7 +79,7 @@ fn v2_and_v3_are_untouched() { let key = day_key(DAY); let cache = Cache::fill_log2(key, 16); // the version 2 item restated by hand (the mixer_mult_by_hand test of memhard.rs, m = 1) - let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0 }); + let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0, state: false }); let t = 12_345u32; let mut s = [0u32; 16]; s[..8].copy_from_slice(&key); @@ -96,7 +96,7 @@ fn v2_and_v3_are_untouched() { mixer(&mut s, round_key(8), &v2); assert_eq!(derive_item(t, &v2, &cache), s); // the mixer constants of the derivation class are the v2 draws (the stream continues after them) - let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8 }); + let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false }); assert_eq!((dr.rot, dr.mul, dr.rc), (v2.rot, v2.mul, v2.rc)); } @@ -109,12 +109,12 @@ fn stream_class_name_and_id() { rng.next(); } let expect = DeriveProgram::draw(&mut rng, DERIVE_LEN_X8); - let mp = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8 }); + let mp = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8, state: false }); assert_eq!(mp.derive.as_ref().unwrap(), &expect); // another day, another program; another length, another program - let other = MixParams::with_shape(day_key("2026-10-04"), Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8 }); + let other = MixParams::with_shape(day_key("2026-10-04"), Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8, state: false }); assert_ne!(other.derive.as_ref().unwrap().fingerprint(), expect.fingerprint()); - let short = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: 368 }); + let short = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: 368, state: false }); assert_eq!(short.derive.as_ref().unwrap().instr_count(), 9 * 368); // the class: name, parse, id, and the v2 program stream (v2 loads, no width roll) let c = LoadClass::DR736; @@ -179,8 +179,8 @@ fn determinism_and_pack_text() { fn stats_beside_x8() { let key = day_key(DAY); let cache = Cache::fill_log2(key, 18); - let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 18, derive_len: DERIVE_LEN_X8 }); - let x8 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 18, derive_len: 0 }); + let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 18, derive_len: DERIVE_LEN_X8, state: false }); + let x8 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 18, derive_len: 0, state: false }); for (label, mp) in [("dr736", &dr), ("x8", &x8)] { let n = 2048u32; let mut ones = [0u32; 512]; @@ -265,7 +265,7 @@ fn text_forms_match_scalar_reference() { /// The dataset source of the class on a day: the verifier's `word` path derives through the program. #[test] fn dataset_source_word_path() { - let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8 }); + let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false }); let m = ds.memhard().unwrap(); let item = derive_item(3, &m.params, &m.cache); for j in 0..16u32 { diff --git a/igneum-pow/tests/mixer.rs b/igneum-pow/tests/mixer.rs index eebda7f47..1a78852bb 100644 --- a/igneum-pow/tests/mixer.rs +++ b/igneum-pow/tests/mixer.rs @@ -254,7 +254,7 @@ fn edge_items_every_multiplier() { let key = day_key(DAY); let cache = Cache::fill_log2(key, 14); for m in [1u32, 2, 4, 8] { - let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 14, derive_len: 0 }); + let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 14, derive_len: 0, state: false }); let by_hand = |t: u32| -> [u32; 16] { let mut s = [0u32; 16]; s[..8].copy_from_slice(&key); diff --git a/igneum-pow/tests/packs.rs b/igneum-pow/tests/packs.rs index 98446833d..a7ce99c5c 100644 --- a/igneum-pow/tests/packs.rs +++ b/igneum-pow/tests/packs.rs @@ -369,7 +369,7 @@ fn v3_packs_are_the_v2_seeds_under_mixer_x8() { assert_eq!(e3.program.program_id(), igneum_pow::generator::program_id(GENERATOR_VERSION_V3, &e3.program.seed, e3.program.attempt)); let m3 = e3.dataset.memhard().unwrap(); let m2 = e2.dataset.memhard().unwrap(); - assert_eq!(m3.shape(), Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0 }); + assert_eq!(m3.shape(), Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: false }); assert_eq!(m3.cache.fnv1a64(), m2.cache.fnv1a64(), "{v3}: the same cache as v2 on day 0"); assert_eq!(m3.params.rot, m2.params.rot); assert_eq!(e3.dataset.log2_words, 28); @@ -405,7 +405,7 @@ fn v3_packs_are_the_v2_seeds_under_mixer_x8() { assert_eq!(j["load_class"].as_str().unwrap(), "mx4"); assert_eq!(e4.program.class, LoadClass::MX4); assert_eq!(e4.program.instrs, epoch(v2).program.instrs); - assert_eq!(e4.dataset.memhard().unwrap().shape(), Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0 }); + assert_eq!(e4.dataset.memhard().unwrap().shape(), Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0, state: false }); assert!(read(x4, "memhard.h").contains("j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u))")); } }