From ae710a39909410ed032a62870eb7b3c6dcfc475f Mon Sep 17 00:00:00 2001 From: igneum-labs <337424239+igneum-labs@users.noreply.github.com> Date: Thu, 8 Oct 2026 21:30:18 +0000 Subject: [PATCH] verdict-cache-fix: igneum-pow back to master's tree (the release-2.0.0 freeze pairing for box node builds had been staged by a checkout and rode into 9b8a31a9a; the spec read-back read it red) Co-Authored-By: Claude Fable 5.1 --- igneum-pow/src/accept.rs | 306 +++++++++++++++++- igneum-pow/src/bind.rs | 15 +- igneum-pow/src/emit.rs | 559 ++++++++++++++++++++++++++++---- igneum-pow/src/generator.rs | 619 ++++++++++++++++++++++++++++++++---- igneum-pow/src/main.rs | 86 ++++- igneum-pow/src/memhard.rs | 93 ++++-- igneum-pow/src/packcheck.rs | 6 +- igneum-pow/src/verify.rs | 367 +++++++++++++++++++-- igneum-pow/tests/mixer.rs | 4 +- igneum-pow/tests/packs.rs | 303 +++++++++++++++++- 10 files changed, 2158 insertions(+), 200 deletions(-) diff --git a/igneum-pow/src/accept.rs b/igneum-pow/src/accept.rs index 9937dba06..f96ab824b 100644 --- a/igneum-pow/src/accept.rs +++ b/igneum-pow/src/accept.rs @@ -111,6 +111,13 @@ pub enum Reject { HotItemSite { site: u8, distinct: u32, evaluations: u32, ratio_milli: u32, top_index: u32, top_count: u32 }, /// (c): output bit `bit` was set in `ones` of 2,048 hashes. OutputBias { bit: u8, ones: u32 }, + /// The reg64 window's liveness rule (the coordinator's spec of 8 October 2026, `check_window_liveness`, a research + /// class, not wired into the acceptance): xoring register `reg` of lane `lane` with a probe word at the start of iteration 0 + /// left the iteration's first load address unchanged (`address_changed` false) or the final hash unchanged + /// (`result_changed` false). A window whose address fold reads a subset of the registers is refused here. + DeadWindowRegister { reg: u8, lane: u8, address_changed: bool, result_changed: bool }, + /// `check_window_liveness` on a program without the reg64 window. + NotAWindow, /// (c): the distinct-address sum was `sum`. DistinctAddresses { sum: u64 }, } @@ -133,6 +140,13 @@ impl std::fmt::Display for Reject { Reject::LowEntropySite { site, distinct, evaluations, ratio_milli } => write!(f, "(c'') load site {site} read {distinct} distinct word indices over {evaluations} evaluations, {}.{:03} of a uniform source on its window (floor {MIN_DISTINCT_RATIO_V4} at 2^20)", ratio_milli / 1000, ratio_milli % 1000), Reject::SaturatedSource { site, count } => write!(f, "(c') load site {site} read a saturated source value in {count} of 16384 evaluations (limit 163)"), Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"), + Reject::DeadWindowRegister { reg, lane, address_changed, result_changed } => write!( + f, + "(reg64 liveness) xoring register {reg} of lane {lane} with a probe word at the start of iteration 0 left the first load address {} and the result {}", + if *address_changed { "changed" } else { "unchanged" }, + if *result_changed { "changed" } else { "unchanged" } + ), + Reject::NotAWindow => write!(f, "(reg64 liveness) not a reg64 window class"), Reject::DistinctAddresses { sum } => { write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0) } @@ -213,6 +227,116 @@ pub fn check_distinct_indices_v4(p: &Program) -> Result<(), Reject> { distinct_ratio_pass(p, ACCEPT_UNITS_DISTINCT_V4, MIN_DISTINCT_RATIO_V4).map(|_| ()) } +/// The index-bit bias read of class v6 lane 1 (`docs/analysis/class-v6/family-gate.md` section 3.2, the value-level +/// test of adv-cache-2; `docs/design/class-v6-rotating-family.md` section 5.1): a research-class check beside (c''), +/// NOT wired into the acceptance (as a refusal it would redraw about 40 percent of epochs on the plain `load_index`; +/// with the index fold in it is the guard that the fold holds). Per load site, in load order, the one-count of every +/// address bit over the site's `units x 32 x 8` word indices on the rule's closed form (`2^28` words) against n / 2, +/// in sigma (`sigma = sqrt(n) / 2`, 512 at the (c'') sample of 2^20). The free bits of a site are those at or above +/// the width's alignment and below the window's cut (`28 - k`, `k = min(win, 2)`); a bit the window fixes reads 0 or +/// n by construction and is reported as 0 sigma. `Err` when the run hits a lane-constant site. The known-failed set is +/// lane D's: on the plain address, F8's p4, p8, p10, p15, p34 (class v4) and the gate's p212, p225 (class v5) each +/// carry one site over 6 sigma at address bit R or R + 1 (the era's rotation); with the fold every one reads under 6. +pub fn index_bit_sigma(p: &Program, units: usize) -> Result, Reject> { + const BITS: usize = ACCEPT_DATASET_LOG2 as usize; + let indices = site_indices(p, units)?; + let n = (units * LANES * ITERATIONS) as f64; + let sigma = n.sqrt() / 2.0; + let mut out = Vec::with_capacity(indices.len()); + let loads: Vec<&Instr> = p.instrs.iter().filter(|i| i.op.is_load()).collect(); + for (site, ix) in indices.iter().enumerate() { + let ins = loads[site]; + let k = (ins.win as u32).min(ACCEPT_DATASET_LOG2.saturating_sub(26)); + let low = (ins.width as u32).trailing_zeros(); + let high = ACCEPT_DATASET_LOG2 - k; + let mut ones = [0u64; BITS]; + for &x in ix { + for (b, o) in ones.iter_mut().enumerate() { + *o += ((x >> b) & 1) as u64; + } + } + let mut z = [0.0f64; BITS]; + for b in 0..BITS { + if (b as u32) >= low && (b as u32) < high { + z[b] = (ones[b] as f64 - n / 2.0) / sigma; + } + } + out.push(z); + } + Ok(out) +} + +/// The largest `|sigma|` of [`index_bit_sigma`] with its site and bit: `(sigma, site, bit)`, signed as read. +pub fn index_bit_sigma_max(p: &Program, units: usize) -> Result<(f64, usize, usize), Reject> { + let z = index_bit_sigma(p, units)?; + let mut best = (0.0f64, 0usize, 0usize); + for (site, row) in z.iter().enumerate() { + for (bit, &v) in row.iter().enumerate() { + if v.abs() > best.0.abs() { + best = (v, site, bit); + } + } + } + Ok(best) +} + +/// The reg64 window's liveness rule (the coordinator's spec, 8 October 2026; a research class, beside the census, +/// NOT wired into the acceptance): the window's state must stay live across the dependent memory chain, not only +/// inside an arithmetic block. For each of the 64 registers in turn, the register is xored with each of two +/// seed-derived pseudo-random words ([`REG64_PROBE_WORDS`]) in every lane of unit 0 (the seed's first acceptance base nonce) at the start of iteration 0, on the closed-form dataset of +/// the acceptance; the program passes when, in every lane, the iteration's first load reads a different word index +/// and the final hash differs. A window whose load addresses read a subset of the registers (the arithmetic-only +/// form, where a load's address is its own window's register) is refused at the first register outside that +/// subset; the full-chain form (every address source `src ^ m`, `m` the rotate-xor chain over the 63 other +/// registers) passes, both patterns moving the address and the result in every lane. Two pseudo-random words, +/// not the complement and not one bit: the chain is linear (xor and rotate), so a pattern that enters it twice +/// through a copy made by the program (the pinned draw's instruction 1, `r15 ^= r8`) cancels when the two copies +/// agree after their rotations, which the complement always does (all ones under any rotation) and a single bit +/// does whenever the rotations agree mod 32; a one-bit flip is also lost through a carry before the first load. +/// A dead register fails both words always; a live one fails both with probability about 2^-60. A program without the window is `Reject::NotAWindow`. +pub const REG64_PROBE_WORDS: usize = 2; + +/// The two probe words of register `reg` under the program's seed: splitmix32 of the seed's first word, the +/// register and the word index, never 0, never all ones, never a single bit (the patterns a linear fold can lose). +pub fn reg64_probe_words(seed: &[u32; 8], reg: usize) -> [u32; REG64_PROBE_WORDS] { + let mut out = [0u32; REG64_PROBE_WORDS]; + for (j, w) in out.iter_mut().enumerate() { + let mut x = splitmix32(seed[0] ^ (reg as u32).wrapping_mul(0x9e3779b9) ^ (j as u32 + 1).wrapping_mul(0x85ebca6b)); + while x == 0 || x == u32::MAX || x.is_power_of_two() { + x = splitmix32(x.wrapping_add(0x6c62272e)); + } + *w = x; + } + out +} + +pub fn check_window_liveness(p: &Program) -> Result<(), Reject> { + use crate::verify::{interpret_warp_probe, Probe}; + if !p.class.reg64 { + return Err(Reject::NotAWindow); + } + let shape = crate::memhard::Shape::for_class(&p.class); + let ds = crate::verify::DatasetSource::from_key_shape(p.seed, crate::verify::DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2, shape); + let base = accept_base_nonces_n(&p.seed, 1)[0]; + let mut plain = Probe::default(); + let plain_out = interpret_warp_probe(p, &p.seed, base, &ds, &mut plain); + assert!(plain.seen_first_load, "a program has a load in every iteration"); + for reg in 0..p.registers() { + for word in reg64_probe_words(&p.seed, reg) { + let mut pr = Probe { flip: Some((reg, word)), ..Default::default() }; + let out = interpret_warp_probe(p, &p.seed, base, &ds, &mut pr); + for lane in 0..LANES { + let address_changed = pr.first_load_idx[lane] != plain.first_load_idx[lane]; + let result_changed = out.hashes[lane] != plain_out.hashes[lane]; + if !address_changed || !result_changed { + return Err(Reject::DeadWindowRegister { reg: reg as u8, lane: lane as u8, address_changed, result_changed }); + } + } + } + } + Ok(()) +} + /// One ratio pass over `units`: every load site's distinct word indices against the uniform expectation on its /// window (`N - N^2 / 2W`, the window `2^28 >> min(win, 2)` words of the closed-form dataset), `Err` at the first /// site under `floor`, else the minimum ratio and its site. @@ -244,10 +368,10 @@ pub fn site_window_words(ins: &Instr) -> u64 { (1u64 << ACCEPT_DATASET_LOG2) >> (ins.win as u64).min(2) } -/// One interpreter run over `units` units with every load site's word indices kept, then per site the sorted -/// run lengths: [`SiteIndexStats`] per site in load order. The ratio pass (c'') and the hot-item rule (c''') read -/// the same run, so class v5 pays the sort once. -pub fn site_index_stats(p: &Program, units: usize) -> Result, Reject> { +/// One interpreter run over `units` units of the seed's acceptance stream with every load site's word indices kept, +/// in evaluation order: one vector per site in load order (the ratio pass sorts them; the index-bit read of class v6 +/// lane 1 counts bits over them). +pub fn site_indices(p: &Program, units: usize) -> Result>, Reject> { let loads = p.loads_per_hash(); let sites = loads / ITERATIONS; let mut acc = Acc { @@ -264,8 +388,16 @@ pub fn site_index_stats(p: &Program, units: usize) -> Result for (unit, &base) in accept_base_nonces_n(&p.seed, units).iter().enumerate() { run_unit(p, unit, base, &mut acc, &mut lane_addrs)?; } - let mut out = Vec::with_capacity(sites); - for ix in acc.indices.take().unwrap().iter_mut() { + Ok(acc.indices.take().unwrap()) +} + +/// One interpreter run over `units` units with every load site's word indices kept, then per site the sorted +/// run lengths: [`SiteIndexStats`] per site in load order. The ratio pass (c'') and the hot-item rule (c''') read +/// the same run, so class v5 pays the sort once. +pub fn site_index_stats(p: &Program, units: usize) -> Result, Reject> { + let mut indices = site_indices(p, units)?; + let mut out = Vec::with_capacity(indices.len()); + for ix in indices.iter_mut() { ix.sort_unstable(); let mut st = SiteIndexStats { distinct: 0, pairs: 0, top_index: 0, top_count: 0 }; let mut i = 0; @@ -354,8 +486,14 @@ pub fn check_indices_v5(p: &Program) -> Result<(), Reject> { /// the era set aside): the shape the sub-version 2 rules (a') and (c') apply to, on every draw path. pub fn is_class_v4_shape(class: &LoadClass) -> bool { matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. })) - // class v5 (docs/design/class-v5-stored-state.md) is judged under the same rules: its state flag is set aside - && LoadClass { era: None, shadow: None, state: false, ..*class } == LoadClass { shadow: None, ..V4_CLASS } + // class v5 (docs/design/class-v5-stored-state.md) is judged under the same rules: its state flag is set aside; + // class v6 lane 1's index fold and re-weight table are set aside too (the address path and the op table are + // not the shape) + // the window and layer 8 off are set aside too (lane D's finding of 8 October 2026, 19:1x UK: a +nowin or +reg64c + // program was judged by the class v2 parts alone on the chain's path; the harness's own predicate hid it) + // and the era's drawn width (lane D's second finding, 20:0x UK: an era whose width set has more than one width + // writes the drawn width into `mix`, and the rule must judge every width of the family, not the pinned one alone) + && LoadClass { era: None, shadow: None, state: false, fold: false, rw: 0, nowin: false, reg64: false, reg64_chain: false, mix: V4_CLASS.mix, ..*class } == LoadClass { shadow: None, ..V4_CLASS } } /// One pass of the dataflow freshness over the base program then the shadow block (the order of one iteration), @@ -715,6 +853,12 @@ pub fn check_dynamic(p: &Program) -> Result { check_distinct_indices_v4(p)?; } } + // the reg64 window's liveness rule (the window's acceptance tool, wired 8 October 2026, 19:5x UK): keyed on the + // window flag so no other class's verdict moves; the arithmetic-only window is refused at its first dead register, + // the full chain passes (129 one-warp interpretations on the closed form, under a second) + if p.class.reg64 { + check_window_liveness(p)?; + } let half = (ACCEPT_HASHES / 2) as u32; let mut bias_max = 0u32; for (bit, &ones) in acc.bit_ones.iter().enumerate() { @@ -883,6 +1027,152 @@ mod tests { assert_eq!(v5_under, 0); } + /// The (c''') reading on a listed set of chain-shaped programs (run by hand on a box through the lease): the file + /// `IGNEUM_V5_PROGRAMS` holds one program per line, ` [tag...]` + /// (a hex field of 64 characters is the 32 bytes; anything else is a label whose `seed_words_from_bytes` words are + /// the bytes, as the f8 and adv label spaces derive theirs). Per line: the class v4 program at that attempt (its id, + /// minimum site and ratio at the 2^20 sample), the class v5 verdict at the same attempt, and the class v5 draw's + /// attempt. The coordinator's question of 7 October 2026, 23:3x UK: whether the drawn-era programs adv-cache-2 read + /// as biased (R from 3 to 22) sit under the 0.995 floor. + #[test] + #[ignore] + fn v5_floor_on_listed_programs() { + use crate::generator::{V5_CLASS, V3_ALLOWED}; + use crate::seed::seed_words_from_bytes; + let path = std::env::var("IGNEUM_V5_PROGRAMS").expect("IGNEUM_V5_PROGRAMS="); + let text = std::fs::read_to_string(&path).expect("the programs file"); + let bytes_of = |f: &str| -> Vec { + if f.len() == 64 && f.chars().all(|c| c.is_ascii_hexdigit()) { + crate::bind::unhex(f).unwrap() + } else { + seed_words_from_bytes(f.as_bytes()).iter().flat_map(|x| x.to_le_bytes()).collect() + } + }; + let mut refused = 0; + let mut rows = 0; + for line in text.lines() { + let line = line.trim(); + if line.is_empty() || line.starts_with('#') { + continue; + } + let f: Vec<&str> = line.split_whitespace().collect(); + if f.len() < 3 { + println!("skip (needs epoch era attempt): {line}"); + continue; + } + let (epoch, era) = (bytes_of(f[0]), bytes_of(f[1])); + // "draw" = the class v4 chain draw's own attempt for this seed (the attempt adv-cache-2's table shows) + let attempt = if f[2] == "draw" { f8_draw(&epoch, &era, V4_CLASS).attempt } else { f[2].parse::().expect("attempt") }; + let tag = f[3..].join(" "); + let label = format!("igneum-epoch/{}", epoch.iter().map(|b| format!("{b:02x}")).collect::()); + let v4 = candidate_class(&label, &epoch, attempt, LoadClass::era(V4_CLASS, &era, &V3_ALLOWED)); + let st = site_index_stats(&v4, ACCEPT_UNITS_DISTINCT_V4).unwrap(); + let (min, site) = distinct_ratio_on(&v4, &st, ACCEPT_UNITS_DISTINCT_V4, 0.0).unwrap(); + let v4_verdict = check(&v4).is_ok(); + let v5 = candidate_class(&label, &epoch, attempt, LoadClass::era(V5_CLASS, &era, &V3_ALLOWED)); + let v5_verdict = check(&v5); + let drawn = f8_draw(&epoch, &era, V5_CLASS); + let era_params = v4.class.era.as_ref().map(|e| e.stride_rot).unwrap_or(0); + rows += 1; + if v5_verdict.is_err() { + refused += 1; + } + println!( + "listed: id {:016x} attempt {attempt} R {era_params} | v4 {} min site {site} ratio {min:.4} (most read {:#x} x{}) | v5 at this attempt {} | v5 draw attempt {} id {:016x} | {tag}", + crate::generator::program_id(crate::generator::GENERATOR_VERSION_V4, &v4.seed, attempt), + if v4_verdict { "accepted" } else { "REJECTED" }, + st[site].top_index, + st[site].top_count, + match &v5_verdict { Ok(_) => "accepted".to_string(), Err(r) => format!("REFUSED {r}") }, + drawn.attempt, + drawn.program_id() + ); + } + println!("listed: {refused} of {rows} refused under class v5 at the listed attempt"); + } + + /// Class v5's attempts census (run by hand on a box through the lease: `IGNEUM_V5_CENSUS_SEEDS` seeds of the f8 label + /// space from `IGNEUM_V5_CENSUS_FROM`, `IGNEUM_V5_CENSUS_THREADS` threads): every candidate of the class v5 chain draw + /// with the FIRST failing part of the rule, so spec 1.4.7's census paragraph states class v5's own per-candidate + /// rejection and its split by part ((a), (b), (a'), (c), (c'), (c''), (c''')), the exhaustions (must be 0) and + /// P(256 consecutive rejections) = r^256, in the form the crypto lane's attempts census gives for sub-version 3 + /// (per-candidate rejection 0.68, (a') 83.3 percent of rejections). + #[test] + #[ignore] + fn v5_attempts_census() { + use crate::generator::{V5_CLASS, V3_ALLOWED}; + use std::sync::atomic::{AtomicU32, Ordering}; + let seeds: u32 = std::env::var("IGNEUM_V5_CENSUS_SEEDS").ok().and_then(|v| v.parse().ok()).unwrap_or(1000); + let from: u32 = std::env::var("IGNEUM_V5_CENSUS_FROM").ok().and_then(|v| v.parse().ok()).unwrap_or(0); + let threads: usize = std::env::var("IGNEUM_V5_CENSUS_THREADS").ok().and_then(|v| v.parse().ok()).unwrap_or(48); + let next = AtomicU32::new(from); + let t0 = std::time::Instant::now(); + let part_of = |r: &Reject| -> &'static str { + match r { + Reject::StaleLoadSource { .. } => "(a) stale load", + Reject::NoInjectingWrite { .. } => "(b) no injecting write", + Reject::UnfreshLoadSource { .. } => "(a') unfresh source", + Reject::ConstantBit { .. } => "(c) constant bit", + Reject::LaneConstantSite { .. } => "(c) lane-constant site", + Reject::Saturated { .. } => "(c) saturated", + Reject::OutputBias { .. } => "(c) output bias", + Reject::DeadWindowRegister { .. } => "(reg64 liveness) dead window register", + Reject::NotAWindow => "(reg64 liveness) not a window", + Reject::DistinctAddresses { .. } => "(c) distinct addresses", + Reject::SaturatedSource { .. } => "(c') saturated source", + Reject::RepeatedSource { .. } => "(c'') repeated source", + Reject::LowEntropySite { .. } => "(c'') low-entropy site", + Reject::HotItemSite { .. } => "(c''') hot-item site", + } + }; + // per seed: (k, accepted attempt or None, the parts of every rejected candidate) + let rows: Vec<(u32, Option, Vec<&'static str>)> = std::thread::scope(|sc| { + let hs: Vec<_> = (0..threads).map(|_| sc.spawn(|| { + let mut out = Vec::new(); + loop { + let k = next.fetch_add(1, Ordering::Relaxed); + if k >= from + seeds { + break out; + } + let (epoch, era) = f8_seed(k); + let class = LoadClass::era(V5_CLASS, &era, &V3_ALLOWED); + let label = f8_label(&epoch); + let mut parts = Vec::new(); + let mut accepted = None; + for attempt in 0..crate::generator::MAX_ATTEMPTS_V4 { + let c = candidate_class(&label, &epoch, attempt, class); + match check(&c) { + Ok(_) => { accepted = Some(attempt); break; } + Err(r) => parts.push(part_of(&r)), + } + } + out.push((k, accepted, parts)); + } + })).collect(); + let mut rows: Vec<_> = hs.into_iter().flat_map(|h| h.join().unwrap()).collect(); + rows.sort_by_key(|r| r.0); + rows + }); + let rejected: usize = rows.iter().map(|r| r.2.len()).sum(); + let candidates = rejected + rows.iter().filter(|r| r.1.is_some()).count(); + let exhausted = rows.iter().filter(|r| r.1.is_none()).count(); + let r = rejected as f64 / candidates as f64; + let mut by_part = std::collections::BTreeMap::new(); + for row in &rows { + for p in &row.2 { + *by_part.entry(*p).or_insert(0usize) += 1; + } + } + println!("v5_attempts_census: {} seeds {from}..{} in {:.0} s on {threads} threads", rows.len(), from + seeds, t0.elapsed().as_secs_f64()); + println!("candidates {candidates}, rejected {rejected}, per-candidate rejection {r:.4}, exhaustions {exhausted}, P(256 consecutive) {:.3e}", r.powf(256.0)); + for (p, n) in &by_part { + println!(" first failing part {p}: {n} ({:.1} percent of rejections, {:.1} percent of candidates)", *n as f64 * 100.0 / rejected as f64, *n as f64 * 100.0 / candidates as f64); + } + let mean = rows.iter().filter_map(|r| r.1).map(|a| a as f64).sum::() / rows.iter().filter(|r| r.1.is_some()).count() as f64; + println!("accepted attempt mean {mean:.3}"); + assert_eq!(exhausted, 0, "no class v5 seed exhausts its attempts"); + } + #[test] fn distinct_bound_scales_with_the_load_count() { assert_eq!(min_distinct_sum(128), MIN_DISTINCT_SUM); diff --git a/igneum-pow/src/bind.rs b/igneum-pow/src/bind.rs index 94edea2bd..dd18b3069 100644 --- a/igneum-pow/src/bind.rs +++ b/igneum-pow/src/bind.rs @@ -91,10 +91,21 @@ pub fn hex(bytes: &[u8]) -> String { /// Bytes from hex (either case). `None` on odd length or a bad digit. pub fn unhex(s: &str) -> Option> { - if s.len() % 2 != 0 { + // on bytes, never on char boundaries (the fuzz of 8 October 2026 found a multi-byte character inside the text + // panicked the slice; a non-ASCII byte is a bad digit) + let b = s.as_bytes(); + if b.len() % 2 != 0 { return None; } - (0..s.len()).step_by(2).map(|i| u8::from_str_radix(&s[i..i + 2], 16).ok()).collect() + let digit = |c: u8| -> Option { + match c { + b'0'..=b'9' => Some(c - b'0'), + b'a'..=b'f' => Some(c - b'a' + 10), + b'A'..=b'F' => Some(c - b'A' + 10), + _ => None, + } + }; + b.chunks(2).map(|c| Some(digit(c[0])? << 4 | digit(c[1])?)).collect() } impl Epoch { diff --git a/igneum-pow/src/emit.rs b/igneum-pow/src/emit.rs index 90105ba01..13ffe7d88 100644 --- a/igneum-pow/src/emit.rs +++ b/igneum-pow/src/emit.rs @@ -16,25 +16,112 @@ use crate::memhard::{ CHACHA_ROUNDS, CHACHA_SIGMA, HOT_TAG, ITEM_ROUNDS, }; use crate::seed::SplitMix64; -use crate::verify::{window, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2, FOLD_MUL, FOLD_ROT}; +use crate::verify::{window, window32, DatasetGeom, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2, FOLD_MUL, FOLD_ROT, INDEX_FOLD_SHIFT}; /// The index expression of a dataset load (era layout, `docs/plans/era-layout.md` section 1.3). For every class /// without an era it is the lottery hash's `rN & MASK`; for an era program it is the one form /// `((rotl_imm(rN * M, R) & WM) | OFF) & MASK` with the site's window constants at the pack's dataset size. -fn load_index_expr(dialect: CoreDialect, era: Option<&EraParams>, ins: &Instr, a: &str, dataset_log2: u32) -> String { +/// Under the multiply-shift geometry (research class ds55, `--dataset-words`) the AND is the dialect's high +/// multiply by the word count: `mulhi(rN, N)` and `mulhi(((rotl_imm(rN * M, R) & WM32) | OFF32), N)` with the +/// window in the source space (`verify::window32`). +/// +/// Class v6 lane 1 (`EraParams::fold`, `verify::stride`): the rotation's argument becomes `fold16_(rN * M)`, the +/// kernel's `y ^ (y >> 16)` helper, in every dialect; the text of every era without the fold is unchanged. +fn load_index_expr(dialect: CoreDialect, era: Option<&EraParams>, ins: &Instr, a: &str, geom: DatasetGeom) -> String { let mask_name = match dialect { CoreDialect::Metal => "MASK", _ => "mask", }; + if geom.mulshift { + let (mulhi, words) = mulhi_name(dialect); + return match era { + None => format!("{mulhi}({a}, {words})"), + Some(e) => { + let (wm, off) = window32(ins, geom.log2); + format!("{mulhi}(((rotl_imm({}, {}u) & {}) | {}), {words})", stride_product_expr(e, a), e.stride_rot, hex(wm), hex(off)) + } + }; + } + let dataset_log2 = geom.log2; match era { None => format!("{a} & {mask_name}"), Some(e) => { let (wm, off) = window(ins, mask_for(dataset_log2), dataset_log2); - format!("((rotl_imm({a} * {}, {}u) & {}) | {}) & {mask_name}", hex(e.stride_mul), e.stride_rot, hex(wm), hex(off)) + format!("((rotl_imm({}, {}u) & {}) | {}) & {mask_name}", stride_product_expr(e, a), e.stride_rot, hex(wm), hex(off)) } } } +/// The product the era's rotation takes: `rN * M`, or `fold16_(rN * M)` under the index fold of class v6 lane 1. +fn stride_product_expr(e: &EraParams, a: &str) -> String { + if e.fold { + format!("fold16_({a} * {})", hex(e.stride_mul)) + } else { + format!("{a} * {}", hex(e.stride_mul)) + } +} + +/// The `fold16_` helper of a kernel under the index fold (empty for every other program, so every pinned pack keeps +/// its text): `y ^ (y >> 16)`, one xor and one shift on the address path, the same in the three dialects. +fn fold_helper_lines(dialect: CoreDialect, p: &Program) -> String { + if !p.class.era.map(|e| e.fold).unwrap_or(false) { + return String::new(); + } + let (qual, u) = match dialect { + CoreDialect::Metal => ("inline", "uint"), + CoreDialect::Cuda => ("__device__ __forceinline__", "uint32_t"), + CoreDialect::OpenCl => ("static inline", "uint"), + }; + format!( + "// Class v6 lane 1, the index fold (8 October 2026, docs/design/class-v6-rotating-family.md section 2; a research class, NOT the lottery\n\ + // hash): every era load address rotates fold16_(src * STRIDE_MUL) instead of the bare product, y ^ (y >> {}), so no era's rotation lands\n\ + // a biased product bit (bit 0 of an odd product is bit 0 of src; bit 1 is set at 3/8) on an address bit.\n\ + {qual} {u} fold16_({u} y) {{ return y ^ (y >> {}u); }}\n", + INDEX_FOLD_SHIFT, INDEX_FOLD_SHIFT + ) +} + +/// The dialect's high 32-bit multiply and the name of the word-count constant (the multiply-shift geometry). +fn mulhi_name(dialect: CoreDialect) -> (&'static str, &'static str) { + match dialect { + CoreDialect::Metal => ("mulhi", "DS_WORDS"), + CoreDialect::Cuda => ("__umulhi", "IGNEUM_DS_WORDS"), + CoreDialect::OpenCl => ("mul_hi", "IGNEUM_DS_WORDS"), + } +} + +/// The word-count lines of a kernel under the multiply-shift geometry (empty under the mask path, so every pinned +/// pack keeps its text): the define the load expressions read, and the comment that says what moved. +fn ds_words_lines(dialect: CoreDialect, geom: DatasetGeom) -> String { + if !geom.mulshift { + return String::new(); + } + let (mulhi, words) = mulhi_name(dialect); + let mut s = String::new(); + s.push_str(&format!( + "// Research class ds55 (8 October 2026, NOT the lottery hash): the dataset holds {} words ({} items, {} bytes),\n", + geom.words, + geom.items(), + geom.bytes() + )); + s.push_str("// not a power of two. Every load address is the multiply-shift range reduction of spec 01 section 1.13.3,\n"); + s.push_str(&format!("// idx = (src * {words}) >> 32 in 64 bits ({mulhi}), in place of src & mask; the mask argument is not read by a load.\n")); + s.push_str(&format!("#define {words} {}\n", hex(geom.words as u32))); + s +} + +/// The warp-coalesced load's base expression (lever b, `wload`): lane 0's register range-reduced and aligned +/// down to 32 words. +fn wload_base_expr(dialect: CoreDialect, bcast: &str, geom: DatasetGeom) -> String { + if geom.mulshift { + let (mulhi, words) = mulhi_name(dialect); + format!("({mulhi}({bcast}, {words}) & ~31u) + lane") + } else { + let wmask = if dialect == CoreDialect::Metal { "WMASK" } else { "wmask" }; + format!("({bcast} & {wmask}) + lane") + } +} + /// The era lines of program.h (empty without an era). fn era_header_lines(p: &Program) -> String { let Some(e) = p.class.era else { return String::new() }; @@ -51,6 +138,10 @@ fn era_header_lines(p: &Program) -> String { s.push_str(&format!("#define IGNEUM_ERA_STRIDE_ROT {}\n", e.stride_rot)); s.push_str(&format!("#define IGNEUM_ERA_INTERLEAVE {{ {}, {}, {}, {} }}\n", e.pos[0], e.pos[1], e.pos[2], e.pos[3])); s.push_str(&format!("#define IGNEUM_ERA_WINDOWS {}\n", jstr(&era_windows(p)))); + if e.fold { + s.push_str(&format!("// Class v6 lane 1 (research): the product is folded before the rotation, y = src * STRIDE_MUL; y ^= y >> {INDEX_FOLD_SHIFT}; y = rotl(y, STRIDE_ROT).\n")); + s.push_str(&format!("#define IGNEUM_ERA_INDEX_FOLD {INDEX_FOLD_SHIFT}\n")); + } s } @@ -111,7 +202,7 @@ enum WideSource { /// same shape in the three dialects: the vector loads differ (`uint4` pointer on Metal and CUDA, `vload4` on /// OpenCL C 1.2). For `width == 1` the caller emits the lottery hash's one-word form instead. fn wide_load_stmt(dialect: CoreDialect, d: &str, idx: &str, width: u8, src: WideSource, closed: Option<(u32, u32)>) -> String { - debug_assert!(width == 4 || width == 16); + debug_assert!(width == 4 || width == 8 || width == 16); let (u, base_ptr) = match dialect { CoreDialect::Metal => ("uint", "dataset"), CoreDialect::Cuda => ("uint32_t", "ds"), @@ -164,7 +255,11 @@ fn program_class_header_lines(p: &Program) -> String { return String::new(); } let mut s = String::new(); - if p.program_class() == ProgramClass::V5 { + if p.program_class() == ProgramClass::V6 { + s.push_str("// Program class v6 (the Igneum 2.0 D1 object, docs/design/class-v6-rotating-family.md): generator version 6, class v5's\n"); + s.push_str("// stored-state dataset with the index fold, the re-weight table and the 64-register window (the v6 flags in the class\n"); + s.push_str("// string); a worker that runs another class refuses this pack, and a job line names the class it wants (class=v6 era=).\n"); + } else if p.program_class() == ProgramClass::V5 { s.push_str("// Program class v5 (proof of stored state and of following, docs/design/class-v5-stored-state.md): generator version 5,\n"); s.push_str("// class v4 over a dataset whose every item is keyed by the window's execution state (IGNEUM_STATE_* below, leaves.bin);\n"); s.push_str("// a worker that runs another class refuses this pack, and a job line names the class it wants (class=v5 era=).\n"); @@ -224,6 +319,14 @@ fn class_header_lines(p: &Program) -> String { } s.push_str(&format!("#define IGNEUM_LOAD_CLASS {} ", jstr(&p.class.name()))); + if p.class.rw != 0 { + // class v6 lane 1: the op roll's table (program.json "op_weights" has the weights) + let (w, sum) = p.class.nonload_weights(); + s.push_str(&format!("// Class v6 lane 1 (research): the op roll draws from re-weight table rw{} (sum {sum}) in place of the plain table (sum 75): {}. +", p.class.rw, w.iter().map(|(o, n)| format!("{} {n}", o.name())).collect::>().join(", "))); + s.push_str(&format!("#define IGNEUM_OP_WEIGHTS_TABLE {} +", p.class.rw)); + } if p.class.mixer_mult != 1 || p.class.growth { s.push_str(&format!("#define IGNEUM_CLASS_MIXER_MULT {} ", p.class.mixer_mult)); @@ -238,6 +341,13 @@ fn class_header_lines(p: &Program) -> String { s.push_str(&format!("#define IGNEUM_CLASS_DERIVE_LEN {} ", p.class.derive_len)); } + if p.class.reg64 { + s.push_str("// reg64 (the hash lane's measurement, 8 October 2026, a research class): 64 live registers per lane; r8..r63 = r[k & 7] * 0x9e3779b9 + k;\n"); + s.push_str("// each drawn instruction i runs on window 0 (register field + 8 * (i % 4)) then on window 1 (+32); both windows fold into r0..r7 by xor before the hash fold.\n"); + s.push_str("#define IGNEUM_REG64 1\n"); + s.push_str(&format!("#define IGNEUM_REGISTERS {}\n", p.registers())); + s.push_str(&format!("#define IGNEUM_REG64_ADDRESS_MIX {} // 1: every load's address source is src ^ m, m the rotate-xor chain (m = first; m = rotl(m, 1) ^ next) over the 63 registers other than src, in index order (the full chain)\n", p.address_mix() as u8)); + } s.push_str(&format!("#define IGNEUM_LOAD_SLOTS {} ", p.class.load_slots)); s.push_str(&format!("#define IGNEUM_LOAD_MIX {{ {}, {}, {} }} @@ -246,7 +356,7 @@ fn class_header_lines(p: &Program) -> String { s.push_str(&format!("#define IGNEUM_LOAD_WIDTH_COUNTS {{ {}, {}, {} }} // loads of 4, 16, 64 bytes per program ", c[0], c[1], c[2])); s.push_str(&format!("#define IGNEUM_BYTES_PER_HASH {} -", p.bytes_per_hash())); +", p.bytes_per_hash_executed())); s.push_str(&format!("#define IGNEUM_FOLD_ROT {FOLD_ROT} ")); s.push_str(&format!("#define IGNEUM_FOLD_MUL {} @@ -397,8 +507,16 @@ fn shadow_block(p: &Program, dialect: CoreDialect) -> String { )); let ty = if dialect == CoreDialect::Cuda { "uint32_t" } else { "uint" }; s.push_str(&format!(" for ({ty} sh = 0u; sh < {reps}u; ++sh) {{\n")); + let prefix = dialect == CoreDialect::Cuda && p.address_mix() && reg64_prefix(); for (k, ins) in p.shadow.iter().enumerate() { + if prefix { + // F05 prefix form: the shadow's writes move S too (the fold reads the live registers) + s.push_str(&format!(" t_ = {};\n", reg64_term(ins.dst as usize))); + } s.push_str(&format!(" {} // s{k} {}\n", shadow_instr_line(dialect, ins), ins.op.name())); + if prefix { + s.push_str(&format!(" S ^= t_ ^ {};\n", reg64_term(ins.dst as usize))); + } } s.push_str(" }\n"); s @@ -861,23 +979,34 @@ const DS_ELEM_BODY: &str = " x *= 0x9E3779B1u; x ^= x >> 15;\n x += d1;\n /// The Metal hash kernel (`generateMSL`, program.metal). pub fn metal_program(p: &Program, dataset_log2: u32, source: LoadSource) -> String { - metal_program_impl(p, dataset_log2, source, false) + metal_program_impl(p, DatasetGeom::pow2(dataset_log2), source, false) +} + +/// [`metal_program`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`). +pub fn metal_program_geom(p: &Program, geom: DatasetGeom, source: LoadSource) -> String { + metal_program_impl(p, geom, source, false) } /// The header-bound Metal kernel (`program_bound.metal`, serve mode of proto-metal): `igneum_hash_bound` reads its /// init words `I` from `constant uint* initw [[buffer(3)]]` (`bind::block_init_words`) instead of `SEEDW`. Same /// instruction text as `igneum_hash`. Stored dataset only. pub fn metal_program_bound(p: &Program, dataset_log2: u32) -> String { - metal_program_impl(p, dataset_log2, LoadSource::Stored, true) + metal_program_impl(p, DatasetGeom::pow2(dataset_log2), LoadSource::Stored, true) } -fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: bool) -> String { - let mask = mask_for(dataset_log2); +/// [`metal_program_bound`] at a dataset geometry. +pub fn metal_program_bound_geom(p: &Program, geom: DatasetGeom) -> String { + metal_program_impl(p, geom, LoadSource::Stored, true) +} + +fn metal_program_impl(p: &Program, geom: DatasetGeom, source: LoadSource, bound: bool) -> String { + let mask = geom.mask(); let mut s = String::with_capacity(5000); s.push_str("#include \n"); s.push_str("using namespace metal;\n"); s.push('\n'); s.push_str(&format!("#define MASK {}\n", hex(mask))); + s.push_str(&ds_words_lines(CoreDialect::Metal, geom)); s.push_str(&hot_define(p)); s.push_str(&format!("constant uint SEEDW[8] = {{ {} }};\n", join_hex(&p.seed))); s.push('\n'); @@ -888,12 +1017,13 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: s.push_str(" return x;\n"); s.push_str("}\n"); s.push_str("inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31\n"); + s.push_str(&fold_helper_lines(CoreDialect::Metal, p)); s.push_str("inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n"); s.push_str("inline uint ds_elem(uint i, uint d0, uint d1) {\n"); s.push_str(" uint x = i ^ d0;\n"); s.push_str(DS_ELEM_BODY); s.push('\n'); - if p.has_wide() { + if p.has_wide() && !geom.mulshift { s.push_str("#define WMASK (MASK & ~31u)\n\n"); } let mut buffer0 = "device const uint* dataset [[buffer(0)]]"; @@ -929,7 +1059,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); } s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str(®_decl(p, "uint")); if p.has_wide() { s.push_str(" uint lane = gid & 31u;\n"); } @@ -941,13 +1071,14 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: (i + 1) & 7 )); } + s.push_str(®64_init(p)); s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); let era = p.class.era; let word_index = |a: &str, wide: bool, ins: &Instr| -> String { if wide { - format!("(simd_broadcast({a}, 0) & WMASK) + lane") + wload_base_expr(CoreDialect::Metal, &format!("simd_broadcast({a}, 0)"), geom) } else { - load_index_expr(CoreDialect::Metal, era.as_ref(), ins, a, dataset_log2) + load_index_expr(CoreDialect::Metal, era.as_ref(), ins, a, geom) } }; let fetch = |idx: String| -> String { @@ -957,10 +1088,17 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: LoadSource::InlineMemhard(_) => format!("mh_word(cache, {idx})"), } }; - for (k, ins) in p.instrs.iter().enumerate() { + // the drawn program, or its two-window interleaving under reg64 (Program::scheduled, the CUDA text's list); + // reg64 full chain: every load's address source is (rS ^ m), m mixed just before the load (the same chain) + let address_mix = p.address_mix(); + let addr_src = |a: &str| -> String { if address_mix { format!("({a} ^ m)") } else { a.to_string() } }; + for (k, ins) in p.scheduled().iter().enumerate() { let d = format!("r{}", ins.dst); let a = format!("r{}", ins.src); let b = format!("r{}", ins.src2); + if address_mix && ins.op == Op::Load { + s.push_str(&format!(" {}\n", reg64_mix_line(p, ins.src))); + } let line = match ins.op { Op::Add => format!( "{d} = {d} + {a} + select({}, {}, ((sel >> {}u) & 1u) != 0u);", @@ -983,9 +1121,9 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: LoadSource::InlineClosed(d0, d1) => (WideSource::InlineClosed, Some((*d0, *d1))), LoadSource::InlineMemhard(_) => (WideSource::InlineMemhard, None), }; - wide_load_stmt(CoreDialect::Metal, &d, &word_index(&a, false, ins), ins.width, src, closed) + wide_load_stmt(CoreDialect::Metal, &d, &word_index(&addr_src(&a), false, ins), ins.width, src, closed) } - Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&a, false, ins))), + Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&addr_src(&a), false, ins))), Op::WLoad => format!("{d} = {d} ^ {};", fetch(word_index(&a, true, ins))), Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a, p.class.scratch_slot_mask()), Op::Hot => hot_stmt(CoreDialect::Metal, &d, &a), @@ -994,6 +1132,7 @@ fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: } s.push_str(&shadow_block(p, CoreDialect::Metal)); s.push_str(" }\n"); + s.push_str(®64_fold(p)); s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); @@ -1027,14 +1166,130 @@ fn init_line(p: &Program, u: &str, i: usize) -> String { ) } +/// The per-lane register declaration of the kernels (`ty` the dialect's 32-bit word: `uint32_t` in CUDA, `uint` in +/// OpenCL and Metal): r0..r7, or r0..r63 under reg64, with `m` for the full chain. +fn reg_decl(p: &Program, ty: &str) -> String { + if !p.class.reg64 { + return format!(" {ty} r0, r1, r2, r3, r4, r5, r6, r7;\n"); + } + let names: Vec = (0..p.registers()).map(|k| format!("r{k}")).collect(); + let m = if p.address_mix() { + if ty == "uint32_t" && reg64_prefix() { + format!("\n {ty} m, S, p_, t_; // reg64 full chain in the F05 prefix form: S the running xor of the rotated registers, p_ the prefix, t_ the old term") + } else { + format!("\n {ty} m; // reg64 full chain: the address mix of all 64 registers before every load") + } + } else { + String::new() + }; + format!(" {ty} {}; // reg64: 64 live registers per lane (two 32-register windows){m}\n", names.join(", ")) +} + +/// reg64 full chain: the mix statement before a load, `m = r0; m = rotl_imm(m, 1u) ^ r1; ... ^ r63;` over the 63 +/// registers other than the load's source `src` (the verifier's `addr_src`, the same chain), one line; the same +/// text in the three dialects (`rotl_imm` is defined in each). +/// Review B's F05 (8 October 2026): the CUDA texts carry the reg64 address mix in its closed prefix form when this +/// switch is on (`igneum-pow export --reg64-prefix`): `S` is the xor of `a[k] = rotl(r[k], (63 - k) mod 32)` over +/// the 64 registers, kept per lane and updated after every write; a load's source is +/// `r[s] ^ ror(P_s, 1) ^ (S ^ P_s ^ a[s])` with `P_s` the xor of the first `s` terms. The same hash, the same vectors +/// (`verify::reg64_address_source`); a measurement text for the fleet, never the definition. OpenCL and Metal keep +/// the fold. +static REG64_PREFIX: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); + +pub fn set_reg64_prefix(on: bool) { + REG64_PREFIX.store(on, std::sync::atomic::Ordering::Relaxed); +} + +pub fn reg64_prefix() -> bool { + REG64_PREFIX.load(std::sync::atomic::Ordering::Relaxed) +} + +/// `rotl(rK, (63 - k) mod 32)` as text; a rotation of 0 is the register itself (`rotl_imm` takes 1..31). +fn reg64_term(k: usize) -> String { + let n = (63 - k) % 32; + if n == 0 { format!("r{k}") } else { format!("rotl_imm(r{k}, {n}u)") } +} + +/// The prefix-form mix statement before a load of source `src` (the F05 text): `m = ror(P, 1) ^ S ^ P ^ a[src]`. +fn reg64_prefix_mix_line(src: u8) -> String { + let s = src as usize; + let prefix: Vec = (0..s).map(reg64_term).collect(); + let p = if prefix.is_empty() { "0u".to_string() } else { prefix.join(" ^ ") }; + format!("p_ = {p}; m = rotr_var(p_, 1u) ^ S ^ p_ ^ {};", reg64_term(s)) +} + +/// The prefix form's running total after the register init: `S = a[0] ^ ... ^ a[63]`. +fn reg64_prefix_init(p: &Program) -> String { + if !p.class.reg64 || !p.address_mix() || !reg64_prefix() { + return String::new(); + } + let terms: Vec = (0..p.registers()).map(reg64_term).collect(); + format!(" // F05 prefix form: the running xor of every rotated register\n S = {};\n", terms.join(" ^ ")) +} + +fn reg64_mix_line(p: &Program, src: u8) -> String { + let mut s = String::new(); + for k in 0..p.registers() { + if k == src as usize { + continue; + } + if s.is_empty() { + s.push_str(&format!("m = r{k};")); + } else { + s.push_str(&format!(" m = rotl_imm(m, 1u) ^ r{k};")); + } + } + s +} + +/// reg64: r8..r63 from the eight seeded registers, `r[k] = r[k & 7] * 0x9E3779B9 + k` (the verifier's text), the +/// same statements in the three dialects. Empty otherwise. +fn reg64_init(p: &Program) -> String { + if !p.class.reg64 { + return String::new(); + } + let mut s = String::from(" // reg64: the derived registers of both windows\n"); + for k in 8..p.registers() { + s.push_str(&format!(" r{k} = r{} * 0x9e3779b9u + {k}u;\n", k & 7)); + } + s +} + +/// reg64: both windows fold into r0..r7 by xor before the hash fold, the same statements in the three dialects. +/// Empty otherwise. +fn reg64_fold(p: &Program) -> String { + if !p.class.reg64 { + return String::new(); + } + let mut s = String::from(" // reg64: fold the 64 registers into the eight output registers\n"); + for k in 0..8 { + let terms: Vec = (k + 8..p.registers()).step_by(8).map(|j| format!("r{j}")).collect(); + s.push_str(&format!(" r{k} = r{k} ^ {};\n", terms.join(" ^ "))); + } + s +} + /// The instruction lines of the CUDA hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`). -fn cuda_instr_lines(p: &Program, dataset_log2: u32) -> String { +fn cuda_instr_lines(p: &Program, geom: DatasetGeom) -> String { let mut s = String::with_capacity(6000); let era = p.class.era; - for (k, ins) in p.instrs.iter().enumerate() { + // the drawn program, or its two-window interleaving under reg64 (Program::scheduled: the verifier reads the same list) + // reg64 full chain: every load's address source is (rS ^ m), m the rotate-xor mix of all 64 registers computed + // just before the load (the verifier's addr_src, the same chain) + let address_mix = p.address_mix(); + let addr_src = |a: &str| -> String { if address_mix { format!("({a} ^ m)") } else { a.to_string() } }; + for (k, ins) in p.scheduled().iter().enumerate() { let d = format!("r{}", ins.dst); let a = format!("r{}", ins.src); let b = format!("r{}", ins.src2); + let prefix = address_mix && reg64_prefix(); + if address_mix && ins.op == Op::Load { + s.push_str(&format!(" {}\n", if prefix { reg64_prefix_mix_line(ins.src) } else { reg64_mix_line(p, ins.src) })); + } + if prefix { + // the old term of the destination, so S can drop it after the write + s.push_str(&format!(" t_ = {};\n", reg64_term(ins.dst as usize))); + } let line = match ins.op { // Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well. Op::Add => format!( @@ -1053,14 +1308,17 @@ fn cuda_instr_lines(p: &Program, dataset_log2: u32) -> String { Op::Mad => format!("{d} = {a} * {b} + {d};"), Op::Shfl => format!("{d} = {d} ^ __shfl_xor_sync(0xffffffffu, {a}, {});", ins.mask), Op::Load if load_width(ins) > 1 => { - wide_load_stmt(CoreDialect::Cuda, &d, &load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None) + wide_load_stmt(CoreDialect::Cuda, &d, &load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &addr_src(&a), geom), ins.width, WideSource::Stored, None) } - Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2)), - Op::WLoad => format!("{d} = {d} ^ ds[(__shfl_sync(0xffffffffu, {a}, 0) & wmask) + lane];"), + Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &addr_src(&a), geom)), + Op::WLoad => format!("{d} = {d} ^ ds[{}];", wload_base_expr(CoreDialect::Cuda, &format!("__shfl_sync(0xffffffffu, {a}, 0)"), geom)), Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a, p.class.scratch_slot_mask()), Op::Hot => hot_stmt(CoreDialect::Cuda, &d, &a), }; s.push_str(&format!(" {line} // {k} {}\n", ins.op.name())); + if prefix { + s.push_str(&format!(" S ^= t_ ^ {};\n", reg64_term(ins.dst as usize))); + } } s } @@ -1073,6 +1331,11 @@ pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String { /// [`cuda_kernel`] at a dataset size (an era program's window constants are literals of the pack's size; every /// other class ignores it). pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { + cuda_kernel_geom(p, memhard, DatasetGeom::pow2(dataset_log2)) +} + +/// [`cuda_kernel_at`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`). +pub fn cuda_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String { let layout = p.class.layout(); let mut s = String::with_capacity(9000); s.push_str(&generated_by(&p.seed_string)); @@ -1087,6 +1350,7 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3 s.push_str("#include \"memhard.h\"\n"); } s.push('\n'); + s.push_str(&ds_words_lines(CoreDialect::Cuda, geom)); s.push_str(&hot_define(p)); s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n"); s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); @@ -1096,6 +1360,7 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3 s.push_str("}\n"); s.push_str("// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.\n"); s.push_str("__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }\n"); + s.push_str(&fold_helper_lines(CoreDialect::Cuda, p)); s.push_str("// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.\n"); s.push_str("__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n"); s.push_str("__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {\n"); @@ -1162,17 +1427,23 @@ pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u3 s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); } s.push_str(" uint32_t nonce = baseNonce + gid;\n"); - s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str(®_decl(p, "uint32_t")); if p.has_wide() { - s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n"); + s.push_str(" uint32_t lane = threadIdx.x & 31u;\n"); + if !geom.mulshift { + s.push_str(" uint32_t wmask = mask & ~31u;\n"); + } } for i in 0..8 { s.push_str(&init_line(p, "uint32_t", i)); } + s.push_str(®64_init(p)); + s.push_str(®64_prefix_init(p)); s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n")); - s.push_str(&cuda_instr_lines(p, dataset_log2)); + s.push_str(&cuda_instr_lines(p, geom)); s.push_str(&shadow_block(p, CoreDialect::Cuda)); s.push_str(" }\n"); + s.push_str(®64_fold(p)); s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n"); @@ -1272,6 +1543,11 @@ pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { /// [`cuda_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]). pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { + cuda_kernel_bound_geom(p, memhard, DatasetGeom::pow2(dataset_log2)) +} + +/// [`cuda_kernel_bound_at`] at a dataset geometry. +pub fn cuda_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String { let mut s = String::with_capacity(9000); s.push_str(&generated_by(&p.seed_string)); s.push_str( @@ -1288,6 +1564,7 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo s.push_str("#include \n"); s.push_str("#include \"program.h\"\n"); s.push('\n'); + s.push_str(&ds_words_lines(CoreDialect::Cuda, geom)); s.push_str("struct IgneumInitWords { uint32_t w[8]; };\n"); s.push('\n'); s.push_str(&hot_define(p)); @@ -1298,6 +1575,7 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo s.push_str(" return x;\n"); s.push_str("}\n"); s.push_str("__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }\n"); + s.push_str(&fold_helper_lines(CoreDialect::Cuda, p)); s.push_str("__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n"); s.push('\n'); let _ = memhard; // the bound kernel reads the stored dataset in both constructions @@ -1311,9 +1589,12 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); } s.push_str(" uint32_t nonce = baseNonce + gid;\n"); - s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str(®_decl(p, "uint32_t")); if p.has_wide() { - s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n"); + s.push_str(" uint32_t lane = threadIdx.x & 31u;\n"); + if !geom.mulshift { + s.push_str(" uint32_t wmask = mask & ~31u;\n"); + } } for i in 0..8 { s.push_str(&format!( @@ -1322,10 +1603,13 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo (i + 1) & 7 )); } + s.push_str(®64_init(p)); + s.push_str(®64_prefix_init(p)); s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n")); - s.push_str(&cuda_instr_lines(p, dataset_log2)); + s.push_str(&cuda_instr_lines(p, geom)); s.push_str(&shadow_block(p, CoreDialect::Cuda)); s.push_str(" }\n"); + s.push_str(®64_fold(p)); s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n"); @@ -1369,13 +1653,20 @@ pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_lo } /// The instruction lines of the OpenCL hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`). -fn opencl_instr_lines(p: &Program, dataset_log2: u32) -> String { +fn opencl_instr_lines(p: &Program, geom: DatasetGeom) -> String { let mut s = String::with_capacity(6000); let era = p.class.era; - for (k, ins) in p.instrs.iter().enumerate() { + // the drawn program, or its two-window interleaving under reg64 (Program::scheduled, the CUDA text's list); + // reg64 full chain: every load's address source is (rS ^ m), m mixed just before the load (the same chain) + let address_mix = p.address_mix(); + let addr_src = |a: &str| -> String { if address_mix { format!("({a} ^ m)") } else { a.to_string() } }; + for (k, ins) in p.scheduled().iter().enumerate() { let d = format!("r{}", ins.dst); let a = format!("r{}", ins.src); let b = format!("r{}", ins.src2); + if address_mix && ins.op == Op::Load { + s.push_str(&format!(" {}\n", reg64_mix_line(p, ins.src))); + } let line = match ins.op { Op::Add => format!( "{d} = {d} + {a} + ((((sel >> {}u) & 1u) != 0u) ? {} : {});", @@ -1393,10 +1684,10 @@ fn opencl_instr_lines(p: &Program, dataset_log2: u32) -> String { Op::Mad => format!("{d} = {a} * {b} + {d};"), Op::Shfl => format!("{{ uint t_; IGNEUM_SHFL_XOR(t_, {a}, {}u); {d} = {d} ^ t_; }}", ins.mask), Op::Load if load_width(ins) > 1 => { - wide_load_stmt(CoreDialect::OpenCl, &d, &load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None) + wide_load_stmt(CoreDialect::OpenCl, &d, &load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &addr_src(&a), geom), ins.width, WideSource::Stored, None) } - Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2)), - Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[(t_ & wmask) + lane]; }}"), + Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &addr_src(&a), geom)), + Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[{}]; }}", wload_base_expr(CoreDialect::OpenCl, "t_", geom)), Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a, p.class.scratch_slot_mask()), Op::Hot => hot_stmt(CoreDialect::OpenCl, &d, &a), }; @@ -1414,7 +1705,12 @@ pub fn opencl_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { /// [`opencl_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]). pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { - let mut s = opencl_kernel_at(p, memhard, dataset_log2); + opencl_kernel_bound_geom(p, memhard, DatasetGeom::pow2(dataset_log2)) +} + +/// [`opencl_kernel_bound_at`] at a dataset geometry. +pub fn opencl_kernel_bound_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String { + let mut s = opencl_kernel_geom(p, memhard, geom); s.push('\n'); s.push_str( "// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.\n", @@ -1439,11 +1735,11 @@ pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_ s.push_str("#endif\n"); s.push_str(&unit_loop); s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str(®_decl(p, "uint")); s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n"); } else { s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str(®_decl(p, "uint")); s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n"); s.push_str("#if IGNEUM_EXCHANGE == 0\n"); s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); @@ -1453,7 +1749,10 @@ pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_ s.push_str("#endif\n"); } if p.has_wide() { - s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n"); + s.push_str(" uint lane = lid & 31u;\n"); + if !geom.mulshift { + s.push_str(" uint wmask = mask & ~31u;\n"); + } } for i in 0..8 { s.push_str(&format!( @@ -1462,10 +1761,12 @@ pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_ (i + 1) & 7 )); } + s.push_str(®64_init(p)); s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); - s.push_str(&opencl_instr_lines(p, dataset_log2)); + s.push_str(&opencl_instr_lines(p, geom)); s.push_str(&shadow_block(p, CoreDialect::OpenCl)); s.push_str(" }\n"); + s.push_str(®64_fold(p)); s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); @@ -1483,6 +1784,11 @@ pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String { /// [`opencl_kernel`] at a dataset size (see [`cuda_kernel_at`]). pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { + opencl_kernel_geom(p, memhard, DatasetGeom::pow2(dataset_log2)) +} + +/// [`opencl_kernel_at`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`). +pub fn opencl_kernel_geom(p: &Program, memhard: Option<&MixParams>, geom: DatasetGeom) -> String { let layout = p.class.layout(); let mut s = String::with_capacity(14000); s.push_str(&generated_by(&p.seed_string)); @@ -1498,6 +1804,7 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: s.push_str("// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.\n"); s.push_str("#ifndef IGNEUM_GROUP\n#define IGNEUM_GROUP 32\n#endif\n"); s.push_str("#ifndef IGNEUM_EXCHANGE\n#define IGNEUM_EXCHANGE 0\n#endif\n"); + s.push_str(&ds_words_lines(CoreDialect::OpenCl, geom)); s.push_str("#ifdef __OPENCL_VERSION__\n"); s.push_str("#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))\n"); s.push_str("#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]\n"); @@ -1538,6 +1845,7 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: s.push_str("}\n"); s.push_str("// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.\n"); s.push_str("static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }\n"); + s.push_str(&fold_helper_lines(CoreDialect::OpenCl, p)); s.push_str("// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.\n"); s.push_str("static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }\n"); s.push_str("static inline uint ds_elem(uint i, uint d0, uint d1) {\n"); @@ -1612,10 +1920,10 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: s.push_str("#endif\n"); s.push_str(&unit_loop); s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str(®_decl(p, "uint")); } else { s.push_str(" uint nonce = baseNonce + gid;\n"); - s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str(®_decl(p, "uint")); s.push_str("#if IGNEUM_EXCHANGE == 0\n"); s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); s.push_str(" uint xk = 0u;\n"); @@ -1624,15 +1932,20 @@ pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: s.push_str("#endif\n"); } if p.has_wide() { - s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n"); + s.push_str(" uint lane = lid & 31u;\n"); + if !geom.mulshift { + s.push_str(" uint wmask = mask & ~31u;\n"); + } } for i in 0..8 { s.push_str(&init_line(p, "uint", i)); } + s.push_str(®64_init(p)); s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); - s.push_str(&opencl_instr_lines(p, dataset_log2)); + s.push_str(&opencl_instr_lines(p, geom)); s.push_str(&shadow_block(p, CoreDialect::OpenCl)); s.push_str(" }\n"); + s.push_str(®64_fold(p)); s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); @@ -1667,7 +1980,8 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { let key = &ds.key; let dataset_log2 = ds.log2_words; let memhard = ds.memhard().map(|m| &m.params); - let mask = mask_for(dataset_log2); + let geom = ds.geom; + let mask = geom.mask(); let mut s = String::with_capacity(2600); s.push_str(&generated_by(&p.seed_string)); s.push_str("// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.\n"); @@ -1685,12 +1999,27 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!("#define IGNEUM_DAY_BYTES_HEX {}\n", jstr(&hex_bytes(&ds.key_bytes)))); s.push_str(&format!("#define IGNEUM_DAY0 {}\n", hex(key[0]))); s.push_str(&format!("#define IGNEUM_DAY1 {}\n", hex(key[1]))); - s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n")); - s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask))); + if geom.mulshift { + s.push_str(&format!("// Research class ds55 (8 October 2026): a dataset of {} words, not a power of two. IGNEUM_DATASET_LOG2 is floor(log2(words));\n", geom.words)); + s.push_str("// the host allocates IGNEUM_DATASET_WORDS words and builds IGNEUM_DATASET_ITEMS items; IGNEUM_MASK is the last word index (the\n"); + s.push_str("// self-test reads dataset[IGNEUM_MASK]) and is never ANDed: every load is idx = (src * IGNEUM_DATASET_WORDS) >> 32 (spec 01 section 1.13.3).\n"); + s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n")); + s.push_str(&format!("#define IGNEUM_DATASET_WORDS {}u\n", geom.words)); + s.push_str(&format!("#define IGNEUM_DATASET_ITEMS {}u\n", geom.items())); + s.push_str(&format!("#define IGNEUM_DATASET_BYTES {}ull\n", geom.bytes())); + s.push_str("#define IGNEUM_DATASET_MULSHIFT 1\n"); + s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask))); + } else { + s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n")); + s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask))); + } s.push_str("#define IGNEUM_LANES 32\n"); s.push_str(&format!("#define IGNEUM_ITERATIONS {ITERATIONS}\n")); s.push_str(&format!("#define IGNEUM_INSTR_COUNT {INSTR_COUNT}\n")); - s.push_str(&format!("#define IGNEUM_LOADS_PER_HASH {}\n", p.loads_per_hash())); + if p.class.reg64 { + s.push_str("// the executed counts per nonce (review B's V6-02): twice the drawn program's under the 64-register window\n"); + } + s.push_str(&format!("#define IGNEUM_LOADS_PER_HASH {}\n", p.loads_per_hash_executed())); s.push_str(&format!("#define IGNEUM_WIDE_LOADS_PER_HASH {}\n", p.wide_loads_per_hash())); s.push_str(&format!("#define IGNEUM_OP_MIX {}\n", jstr(&p.op_mix()))); s.push_str(&program_class_header_lines(p)); @@ -1820,6 +2149,13 @@ pub fn sample_indices(mask: u32) -> Vec { (0..64).map(|_| (sr.next() as u32) & mask).collect() } +/// [`sample_indices`] at a dataset geometry: the same 64 draws through the geometry's range reduction, so the +/// mask path is [`sample_indices`] exactly and the multiply-shift path samples by the mapping the loads use. +pub fn sample_indices_geom(geom: DatasetGeom) -> Vec { + let mut sr = SplitMix64::new(0x6d68_7361_6d70_6c65); + (0..64).map(|_| geom.reduce(sr.next() as u32)).collect() +} + /// vectors.h (`generateVectorsHeader`). pub fn vectors_header( p: &Program, @@ -1903,14 +2239,15 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { let key = &ds.key; let dataset_log2 = ds.log2_words; let memhard = ds.memhard().map(|m| &m.params); - let mask = mask_for(dataset_log2); + let geom = ds.geom; + let mask = geom.mask(); let mut s = String::with_capacity(14000); s.push_str("{\n"); s.push_str(" \"format\": \"igneum-program-pack-3\",\n"); s.push_str(&format!(" \"generator\": {},\n", p.generator)); s.push_str(&format!(" \"attempt\": {},\n", p.attempt)); s.push_str(&format!(" \"program_id\": {},\n", jhex64(p.program_id()))); - s.push_str(" \"program_id_derivation\": \"FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32\",\n"); + s.push_str(&format!(" \"program_id_derivation\": {},\n", jstr(&p.program_id_derivation()))); s.push_str(&format!( " \"dataset_mode\": {},\n", jstr(if memhard.is_some() { "memory-hard" } else { "closed-form" }) @@ -1924,7 +2261,7 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(" \"registers\": 8,\n"); s.push_str(&format!(" \"iterations\": {ITERATIONS},\n")); s.push_str(&format!(" \"instruction_count\": {INSTR_COUNT},\n")); - s.push_str(&format!(" \"loads_per_hash\": {},\n", p.loads_per_hash())); + s.push_str(&format!(" \"loads_per_hash\": {},\n", p.loads_per_hash_executed())); if p.program_class() != ProgramClass::V2 { s.push_str(&format!(" \"program_class\": {},\n", jstr(p.program_class().name()))); if p.program_class() == ProgramClass::V4 { @@ -1950,6 +2287,18 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { if !p.class.is_v2() { let c = p.width_counts(); s.push_str(&format!(" \"load_class\": {},\n", jstr(&p.class.name()))); + if p.class.rw != 0 { + // class v6 lane 1: the re-weight table the op roll draws from, in draw order, and its sum (the roll's range) + let (w, sum) = p.class.nonload_weights(); + s.push_str(&format!( + " \"op_weights\": {{\"table\": \"rw{}\", \"sum\": {sum}, \"draw_order\": [{}], \"rule\": \"class v6 lane 1 (8 October 2026, a research class): the table replaces the plain ten-family table (sum 75) for the op roll of every base and shadow instruction, roll = below(sum) walked over the weights in draw order; a family at weight 0 is never drawn; shuffle stays at 4 under rw1; the program id carries 'rw/' || table_u8\"}},\n", + p.class.rw, + w.iter().map(|(o, n)| format!("[{}, {n}]", jstr(o.name()))).collect::>().join(", ") + )); + } + if p.class.reg64 { + s.push_str(&format!(" \"reg64\": {{\"registers\": {}, \"variant\": {}, \"address_mix\": {}, \"liveness\": {}, \"statements_per_iteration\": {}, \"loads_per_hash\": {}, \"rule\": \"the hash lane's 64-register window (8 October 2026, a research class, NOT the lottery hash): r0..r7 seeded as today, r[k] = r[k & 7] * 0x9E3779B9 + k for k in 8..63; instruction i of the drawn program runs on window 0 with every register field + 8 * (i % 4), then on window 1 with +32 more; after the 8 iterations r[k] ^= r[k + 8] ^ ... ^ r[k + 56] for k in 0..7, then the hash fold; the draw, the acceptance rule and the dataset are the class's without the flag; address_mix 1 (the full chain): every load's address source is src ^ m, m the rotate-xor chain (m = first; m = rotl(m, 1) ^ next) over the 63 registers other than src in index order, computed before the load; src stays out of the chain so no register's term can cancel its own direct term (liveness rule: xoring any register with either of two seed-derived probe words at the start of an iteration moves the iteration's first load address and the final hash)\"}},\n", p.registers(), jstr(if p.address_mix() { "window, full chain" } else { "window, arithmetic-only" }), p.address_mix() as u8, jstr(&p.reg64_liveness()), p.scheduled().len(), p.scheduled().iter().filter(|i| i.op.is_load()).count() * ITERATIONS)); + } if p.class.derive_len != 0 { s.push_str(&format!(" \"derive_len\": {},\n", p.class.derive_len)); s.push_str(" \"derive\": \"Counter ASIC 3.0 item 2 (6 October 2026, docs/plans/counter-asic-3-derivation.md; a prototype, not class v3): the nine mixer slots of the item derivation each run a straight-line program of derive_len instructions drawn from the day key stream after the 40 mixer draws, four draws per instruction (op roll below(100), destination roll below(15), third-register roll below(14), the immediate next()); every instruction reads the register the previous one wrote (s[0] first) and writes another; twelve forms, each a bijection on the state; the 8 dependent cache reads per item unchanged; the acceptance test of derive.rs (every register written per round program, 8 distinct rotations, the x8 mixer's operation and multiply counts as floors) rejects a draw and the next attempt continues the stream\",\n"); @@ -1962,7 +2311,7 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!(" \"load_slots\": {},\n", p.class.load_slots)); s.push_str(&format!(" \"load_mix_percent_4_16_64\": [{}, {}, {}],\n", p.class.mix[0], p.class.mix[1], p.class.mix[2])); s.push_str(&format!(" \"load_width_counts_4_16_64\": [{}, {}, {}],\n", c[0], c[1], c[2])); - s.push_str(&format!(" \"bytes_per_hash\": {},\n", p.bytes_per_hash())); + s.push_str(&format!(" \"bytes_per_hash\": {},\n", p.bytes_per_hash_executed())); if p.has_scratch() { s.push_str(&format!(" \"scratch_ops_per_hash\": {},\n", p.scratch_ops_per_hash())); s.push_str(&format!(" \"scratch_kib_per_warp\": {},\n", p.class.scratch_kb)); @@ -1979,7 +2328,15 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str(&format!(" \"stride_mul\": {},\n", jhex(e.stride_mul))); s.push_str(&format!(" \"stride_rot\": {},\n", e.stride_rot)); s.push_str(&format!(" \"interleave\": [{}, {}, {}, {}],\n", e.pos[0], e.pos[1], e.pos[2], e.pos[3])); - s.push_str(" \"address\": \"y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words\",\n"); + let y = if e.fold { format!("y = src * stride_mul; y ^= y >> {INDEX_FOLD_SHIFT}; y = rotl(y, stride_rot)") } else { "y = rotl(src * stride_mul, stride_rot)".to_string() }; + if geom.mulshift { + s.push_str(&format!(" \"address\": \"{y}; D = floor(log2(words)); k = min(win, D - 26); v = (y & (0xffffffff >> k)) | ((off & (2^k - 1)) << (32 - k)); idx = (v * words) >> 32 in 64 bits (the window in the source space, then the multiply-shift of spec 01 section 1.13.3); a wide load aligns idx down to W words\",\n")); + } else { + s.push_str(&format!(" \"address\": \"{y}; k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words\",\n")); + } + if e.fold { + s.push_str(&format!(" \"index_fold\": \"class v6 lane 1 (8 October 2026, docs/design/class-v6-rotating-family.md section 2, a research class): the product's low bits are folded before the stride rotation, y ^= y >> {INDEX_FOLD_SHIFT}, so no era's rotation lands a biased product bit (bit 0 of an odd product is bit 0 of src; bit 1 is set at 3/8; lane D's 7 of 16 eras at 130 to 511 sigma) on an address bit; the design names the fold and not its form, so this form is the lane's: one xor and one shift on the address path, the same in verify::stride, the three kernel texts (fold16_) and these vectors; the program id carries 'fold/'\",\n")); + } s.push_str(" \"windows\": \"per instruction, after the width roll: win = below(3), off = low32(next()) & (2^win - 1); used on a load slot (the instruction's win and off fields)\",\n"); s.push_str(" \"dataset_word\": \"dataset[w] = item(t(w))[j(w)]: j(w) gathers the bits of w at the interleave positions, t(w) is w with those bits removed\",\n"); s.push_str(" \"program_id_suffix\": \"'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]\"\n"); @@ -2024,17 +2381,34 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { s.push_str( " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n", ); - s.push_str(" \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n"); - s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\""); + if geom.mulshift { + s.push_str(" \"load\": \"dst = dst ^ dataset[(src * dataset.words) >> 32]\",\n"); + s.push_str(" \"wload\": \"base = ((src of lane 0 * dataset.words) >> 32) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\""); + } else { + s.push_str(" \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n"); + s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\""); + } if p.has_hot() { s.push_str(",\n \"hot\": \"dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)\""); } s.push('\n'); s.push_str(" },\n"); s.push_str(" \"dataset\": {\n"); - s.push_str(&format!(" \"log2_words\": {dataset_log2},\n")); - s.push_str(&format!(" \"bytes\": {},\n", 1u64 << (dataset_log2 as u64 + 2))); - s.push_str(&format!(" \"mask\": {},\n", jhex(mask))); + if geom.mulshift { + s.push_str(&format!(" \"log2_words\": {dataset_log2},\n")); + s.push_str(" \"log2_words_note\": \"floor(log2(words)): the dataset is not a power of two (research class ds55, 8 October 2026); allocate words, build items\",\n"); + s.push_str(&format!(" \"words\": {},\n", geom.words)); + s.push_str(&format!(" \"bytes\": {},\n", geom.bytes())); + s.push_str(&format!(" \"items\": {},\n", geom.items())); + s.push_str(" \"mapping\": \"mulshift\",\n"); + s.push_str(" \"index\": \"idx = (src * words) >> 32 computed in 64 bits (spec 01 section 1.13.3, the multiply-shift range reduction; uniform to within 2^-32, branch-free, integer only), in place of src & mask; the item index is idx >> 4 under the linear layout and t(idx) under an era layout, below items\",\n"); + s.push_str(&format!(" \"mask\": {},\n", jhex(mask))); + s.push_str(" \"mask_note\": \"the last word index, words - 1; never ANDed under the multiply-shift\",\n"); + } else { + s.push_str(&format!(" \"log2_words\": {dataset_log2},\n")); + s.push_str(&format!(" \"bytes\": {},\n", 1u64 << (dataset_log2 as u64 + 2))); + s.push_str(&format!(" \"mask\": {},\n", jhex(mask))); + } s.push_str(&format!(" \"day\": {},\n", jstr(day))); s.push_str(&format!(" \"day_bytes\": {},\n", jstr(&hex_bytes(&ds.key_bytes)))); s.push_str(" \"day_words_from\": \"seed_words_from_bytes(day_bytes)\",\n"); @@ -2178,12 +2552,36 @@ pub fn vectors_json( source: &str, memhard: bool, ) -> String { + let geom = DatasetGeom::pow2(dataset_log2); + debug_assert_eq!(mask, geom.mask()); + vectors_json_geom(p, day, geom, bases, outs, v, source, memhard) +} + +/// [`vectors_json`] at a dataset geometry: under the multiply-shift the file also carries `dataset_words` and +/// `dataset_mapping`, and `dataset_last_index` is `words - 1`. +#[allow(clippy::too_many_arguments)] +pub fn vectors_json_geom( + p: &Program, + day: &str, + geom: DatasetGeom, + bases: &[u32], + outs: &[[u64; 32]], + v: &PackVectors, + source: &str, + memhard: bool, +) -> String { + let dataset_log2 = geom.log2; + let mask = geom.mask(); let mut s = String::with_capacity(6500); s.push_str("{\n"); s.push_str(&format!(" \"seed\": {},\n", jstr(&p.seed_string))); s.push_str(&format!(" \"day\": {},\n", jstr(day))); s.push_str(&format!(" \"dataset_mode\": {},\n", jstr(if memhard { "memory-hard" } else { "closed-form" }))); s.push_str(&format!(" \"dataset_log2_words\": {dataset_log2},\n")); + if geom.mulshift { + s.push_str(&format!(" \"dataset_words\": {},\n", geom.words)); + s.push_str(" \"dataset_mapping\": \"mulshift\",\n"); + } s.push_str(&format!(" \"mask\": {},\n", jhex(mask))); s.push_str(" \"lanes\": 32,\n"); s.push_str(&format!(" \"source\": {},\n", jstr(source))); @@ -2255,15 +2653,17 @@ impl Pack { pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack { let p = &epoch.program; let ds: &DatasetSource = &epoch.dataset; - let mask = ds.mask; + let geom = ds.geom; + let mask = geom.mask(); let memhard = ds.memhard().map(|m| &m.params); let bases = PACK_VECTOR_BASES.to_vec(); let outs: Vec<[u64; 32]> = bases.iter().map(|&b| epoch.hash_warp(b)).collect(); - // the self-test words under the program's layout (era layout; linear for every other class) + // the self-test words under the program's layout (era layout; linear for every other class) and the + // geometry's range reduction (the sampled indices go through the mapping the loads use) let mut v = PackVectors { head: (0..16).map(|i| epoch.dataset_word(i)).collect(), - last: epoch.dataset_word(mask), - sample_idx: sample_indices(mask), + last: epoch.dataset_word(geom.last_index()), + sample_idx: sample_indices_geom(geom), ..Default::default() }; v.sample_val = v.sample_idx.iter().map(|&i| epoch.dataset_word(i)).collect(); @@ -2282,19 +2682,42 @@ pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack { v.hot_fnv = h.fnv1a64(); } let is_mh = memhard.is_some(); + let texts: Vec<(String, String)> = vec![ + ("kernel.cu".to_string(), cuda_kernel_geom(p, memhard, geom)), + ("kernel.cl".to_string(), opencl_kernel_geom(p, memhard, geom)), + ("program.metal".to_string(), metal_program_geom(p, geom, LoadSource::Stored)), + // Header-bound kernels (3 October 2026, bind.rs): new files, the seven above are unchanged. + ("program_bound.metal".to_string(), metal_program_bound_geom(p, geom)), + ("kernel_bound.cu".to_string(), cuda_kernel_bound_geom(p, memhard, geom)), + ("kernel_bound.cl".to_string(), opencl_kernel_bound_geom(p, memhard, geom)), + ]; + // A06 (the external review of 8 October 2026): the pack authenticates its kernel texts by hash, not by metadata: + // identity.json carries the program id, the dataset and era identities and the BLAKE2b-256 of every kernel file. + // A file of its own, so every pinned pack's twelve files stay byte for byte (the readers ignore it). + let hashes: Vec = texts.iter().map(|(n, t)| format!(" {}: \"{}\"", jstr(n), hex_bytes(&crate::blake2b::blake2b_256(&[t.as_bytes()])))).collect(); + let identity = format!( + "{{\n \"program_id\": \"{:#018x}\",\n \"load_class\": {},\n \"generator\": {},\n \"day\": {},\n \"dataset_bytes\": {},\n \"kernel_blake2b256\": {{\n{}\n }},\n \"rule\": \"the external review's A06 (8 October 2026): a pack's program, dataset and work identities are distinct, and its kernel texts are authenticated by the hash of their bytes, never by the metadata beside them\"\n}}\n", + p.program_id(), + jstr(&p.class.name()), + p.generator, + jstr(day), + (ds.geom.words as u128) * 4, + hashes.join(",\n") + ); + // the pack's file order as the pinned packs carry it, identity.json second + let mut texts = texts.into_iter(); + let kernel_cu = texts.next().unwrap(); + let kernel_cl = texts.next().unwrap(); let mut files = vec![ ("program.json".to_string(), program_json(p, day, ds)), - ("vectors.json".to_string(), vectors_json(p, day, ds.log2_words, &bases, &outs, &v, mask, source, is_mh)), - ("kernel.cu".to_string(), cuda_kernel_at(p, memhard, ds.log2_words)), - ("kernel.cl".to_string(), opencl_kernel_at(p, memhard, ds.log2_words)), + ("identity.json".to_string(), identity), + ("vectors.json".to_string(), vectors_json_geom(p, day, geom, &bases, &outs, &v, source, is_mh)), + kernel_cu, + kernel_cl, ("program.h".to_string(), program_header(p, day, ds)), ("vectors.h".to_string(), vectors_header(p, &bases, &outs, &v, mask, source, is_mh)), - ("program.metal".to_string(), metal_program(p, ds.log2_words, LoadSource::Stored)), - // Header-bound kernels (3 October 2026, bind.rs): new files, the seven above are unchanged. - ("program_bound.metal".to_string(), metal_program_bound(p, ds.log2_words)), - ("kernel_bound.cu".to_string(), cuda_kernel_bound_at(p, memhard, ds.log2_words)), - ("kernel_bound.cl".to_string(), opencl_kernel_bound_at(p, memhard, ds.log2_words)), ]; + files.extend(texts); if let Some(mp) = memhard { files.push(("memhard.h".to_string(), cuda_memhard_header(p, mp))); files.push(("memhard.metal".to_string(), metal_memhard_for(p, mp))); diff --git a/igneum-pow/src/generator.rs b/igneum-pow/src/generator.rs index b3667c8c3..ce0084681 100644 --- a/igneum-pow/src/generator.rs +++ b/igneum-pow/src/generator.rs @@ -242,6 +242,44 @@ pub struct LoadClass { /// the item derivation XORs the window's state leaf into every item before the first mixer (`crate::state`, /// `memhard::derive_items_leaves`). The program draw does not read it. `false` for every other class. pub state: bool, + /// W = 8 (the floor programme, 8 October 2026, a research class behind `w32`): the 4-word slot of the width set + /// reads 8 words (one 32-byte sector) instead of 4; the mix, the slots and every other rule stand. `false` for + /// every other class. See [`LoadClass::widths`]. + pub wide8: bool, + /// Class v6 lane 1 (`docs/design/class-v6-rotating-family.md` section 2, the index fold; a research class, + /// 8 October 2026): `true` folds the product's low bits in every era load address before the stride rotation + /// (`verify::load_index`: `y = x * M; y ^= y >> 16; y = rotl(y, R)`), so no era's R lands a biased product bit + /// on an address bit. The class's era carries the same bit ([`EraParams::fold`]). `false` for every other class. + pub fold: bool, + /// Class v6 lane 1, the op-mix re-weight behind the fold: 0 is the plain table [`NONLOAD_WEIGHTS`]; 1 the k lane's + /// optimiser split [`NONLOAD_WEIGHTS_RW`] (sum 83, `or` never drawn); 2 the census lane's neighbouring table + /// [`NONLOAD_WEIGHTS_RW2`] (sum 75). The draw rolls against the table's own sum ([`LoadClass::nonload_weights`]). + pub rw: u8, + /// D2 experiment "layer 8 off" (the coordinator's order, 8 October 2026, 18:12 UK; a research class behind `+nowin`): + /// the era layout's window-layer draw is removed, every load site reads the whole dataset (`win = 0`, `off = 0`). + /// The program stream still consumes the two window draws per instruction, so the base program's instructions are + /// the class's without the flag; only the windows move. `false` for every other class. + pub nowin: bool, + /// The 64-register window (the hash lane's reg64 measurement, 8 October 2026, a research class behind `+reg64` + /// and `--reg64`): each lane holds 64 live 32-bit registers. r0..r7 are seeded as today, r8..r63 derived from them + /// (`r[k] = r[k & 7] * 0x9E3779B9 + k`); the 64 drawn instructions run twice per iteration in an interleaved + /// order, instruction i on window 0 (register field + 8 * (i % 4), registers 0..31) then the same instruction on + /// window 1 (+32, registers 32..63); the windows fold into r0..r7 by xor before the hash fold + /// ([`Program::scheduled`], [`crate::verify`], the emitters). The draw, the acceptance rule and the dataset are the + /// class's without the flag. `false` for every other class. + pub reg64: bool, + /// reg64, the full chain (the coordinator's amendment of 8 October 2026, 15:1x UK, class suffix `+reg64c`, + /// `--reg64-chain`): the address of every load consumes all 64 registers: address source `src ^ m`, `m` the + /// rotate-xor chain (`m = first; m = rotl(m, 1) ^ next`) over the 63 registers other than `src` in index order + /// (the same text in the verifier and the emitters), so an in-flight hash holds 64 independently necessary + /// values for the length of the dependent memory chain. The source stays out of the chain: inside it, r31 and + /// r63 land at rotation 0 mod 32 and cancel their own direct term (found by the liveness rule on the pinned draw). + /// Init rule: r0..r7 from the seed words as every class, `r[k] = r[k & 7] * 0x9E3779B9 + k` for k in 8..63. + /// Output rule: `r[k] ^= r[k + 8] ^ r[k + 16] ^ ... ^ r[k + 56]` for k in 0..7, then the class's hash fold. + /// Liveness rule (`accept::check_window_liveness`): xoring any one register with either of two seed-derived + /// probe words at the start of an iteration moves that iteration's first load address and the final hash, in + /// every lane (the complement and a single bit are the patterns a linear fold loses). Requires `reg64`. + pub reg64_chain: bool, } /// The parameters one era draws from its seed `E_n` (`docs/plans/era-layout.md` section 1.1, the proposed text of @@ -263,6 +301,9 @@ pub struct EraParams { /// Layer 4, interleave: the four ascending bit positions (0..15) of the word-within-item bits in the word /// index; the first `log2(width_words)` are `0..`, so one aligned load stays inside one item. pub pos: [u8; 4], + /// The index fold of class v6 lane 1 ([`LoadClass::fold`]): the product's low bits folded before the rotation. + /// Not drawn: set from the class the era is composed over. `false` for every era of every other class. + pub fold: bool, } /// Domain tag of the era stream seed. @@ -322,7 +363,7 @@ impl EraParams { pub fn era_draw(era_bytes: &[u8], allowed: &[u8]) -> EraParams { assert!(!allowed.is_empty() && allowed.len() <= 3, "the allowed width set has 1 to 3 entries"); for (i, &w) in allowed.iter().enumerate() { - assert!(WIDTH_WORDS.contains(&w), "allowed width {w} is not 1, 4 or 16 words"); + assert!(WIDTH_WORDS.contains(&w) || w == 8, "allowed width {w} is not 1, 4, 8 or 16 words"); assert!(i == 0 || allowed[i - 1] < w, "the allowed width set is ascending"); } let words = EraParams::stream_words(era_bytes); @@ -358,7 +399,7 @@ pub fn era_draw(era_bytes: &[u8], allowed: &[u8]) -> EraParams { } let mut al = [0u8; 3]; al[..allowed.len()].copy_from_slice(allowed); - EraParams { words, allowed: al, width_words, stride_mul, stride_rot, pos } + EraParams { words, allowed: al, width_words, stride_mul, stride_rot, pos, fold: false } } /// The hot table of a class: `mb` MiB (32, 64 or 96 in the experiment) and `k` hot slots. Two forms: `replaced` @@ -419,13 +460,13 @@ impl LoadClass { impl LoadClass { /// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash. pub const V2: LoadClass = - LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None, state: false }; + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None, state: false, wide8: false, fold: false, rw: 0, nowin: false, reg64: false, reg64_chain: false }; /// The construction decided for program class v3 on 5 October 2026 (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): /// version 2 loads (16 slots of one word, no scratch, no width roll, so the program stream is version 2's), the /// mixer applied 4 times per round, and the cache growth rule. Name "mx4". pub const MX4: LoadClass = - LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None, state: false }; + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None, state: false, wide8: false, fold: false, rw: 0, nowin: false, reg64: false, reg64_chain: false }; /// The era class over `base` (`docs/plans/era-layout.md`): the parameters drawn by [`era_draw`]; when `allowed` /// has more than one width the drawn width becomes the class mix (every load that width), otherwise the base @@ -439,12 +480,14 @@ impl LoadClass { c.mix = [0, 0, 0]; c.mix[i] = 100; } else { - let widest = (0..3).rev().find(|&i| c.mix[i] > 0).map(|i| WIDTH_WORDS[i]).unwrap_or(1); + let widest = (0..3).rev().find(|&i| c.mix[i] > 0).map(|i| c.widths()[i]).unwrap_or(1); if widest != e.width_words { // the interleave must keep the widest load inside one item: redraw the positions for that width e = era_draw(era_bytes, &[widest]); } } + // class v6 lane 1: the era's address path carries the class's index fold + e.fold = c.fold; c.era = Some(e); c } @@ -486,6 +529,16 @@ impl LoadClass { LoadClass { mix, load_slots, ..LoadClass::V2 } } + /// The width set this class draws from, in words: [`WIDTH_WORDS`], or `[1, 8, 16]` under `wide8` (W = 8). + pub fn widths(&self) -> [u8; 3] { + if self.wide8 { [1, 8, 16] } else { WIDTH_WORDS } + } + + /// W = 8: every load reads 8 words (a 32-byte sector), `load_slots` loads per program (the class `w32`). + pub fn fixed8(load_slots: u8) -> LoadClass { + LoadClass { mix: [0, 100, 0], load_slots, wide8: true, ..LoadClass::V2 } + } + /// Per-load width drawn from `mix` (percent for 4, 16, 64 bytes), 16 loads per program. pub fn mixed(mix: [u8; 3]) -> LoadClass { assert_eq!(mix.iter().map(|&m| m as u32).sum::(), 100, "the mix must sum to 100"); @@ -553,7 +606,19 @@ impl LoadClass { /// This class with another class's era draw (tests: a rung's class composed with the chain's era). pub fn with_era_of(self, other: &LoadClass) -> LoadClass { - LoadClass { era: other.era, ..self } + // the era's fold bit follows THIS class's flag, not the other's + LoadClass { era: other.era.map(|mut e| { e.fold = self.fold; e }), ..self } + } + + /// The class with the 64-register window per lane ("mx8+reg64", a research class). + pub fn with_reg64(self) -> LoadClass { + LoadClass { reg64: true, ..self } + } + + /// The reg64 class with the full-chain address mix ("mx8+reg64c"). + pub fn with_reg64_chain(self) -> LoadClass { + assert!(self.reg64, "the full-chain address mix is a reg64 variant"); + LoadClass { reg64_chain: true, ..self } } /// The class with the state leaves of class v5 folded into every item ("mx8+sh256x27+state"). @@ -561,6 +626,44 @@ impl LoadClass { LoadClass { state: true, ..self } } + /// Class v6 lane 1: the class with the index fold on every era load address ("...+fold"). An era already + /// composed takes the bit too, so the flag and the era never disagree. + pub fn with_fold(self) -> LoadClass { + LoadClass { fold: true, era: self.era.map(|mut e| { e.fold = true; e }), ..self } + } + + /// Class v6 lane 1: the class drawing its non-load ops from re-weight table `rw` (1: the k lane's split, "+rw"; + /// 2: the census lane's table, "+rw2"; 0: the plain table). + pub fn with_rw(self, rw: u8) -> LoadClass { + assert!(rw <= 2, "re-weight table must be 0, 1 or 2"); + LoadClass { rw, ..self } + } + + /// D2 "layer 8 off": the class with the window-layer draw removed (every load site reads the whole dataset). + pub fn with_nowin(self) -> LoadClass { + LoadClass { nowin: true, ..self } + } + + /// The class v6 object's draw rules (the external review of 8 October 2026, A02 and A08, fixed by construction for + /// every class carrying a v6 flag: the index fold, a re-weight table, the register window or layer 8 off): five + /// of the non-load slots are shuffles whose masks are the five lane dimensions 1, 2, 4, 8, 16 in a drawn order, so + /// every accepted program mixes all 32 lanes by construction; and a `mad` never names its destination as its + /// second source (`d = a * d + d` is `d * (a + 1)`, not a bijection in `d` when `a` is odd). Every other class + /// draws as before. + pub fn is_v6(&self) -> bool { + self.fold || self.rw != 0 || self.reg64 || self.nowin + } + + /// The non-load op table this class draws from, in draw order, and its sum (the roll's range). The plain table + /// for every class without the re-weight flag, so their streams are byte for byte what they were. + pub fn nonload_weights(&self) -> (&'static [(Op, u64); 10], u64) { + match self.rw { + 1 => (&NONLOAD_WEIGHTS_RW, NONLOAD_WEIGHTS_RW_SUM), + 2 => (&NONLOAD_WEIGHTS_RW2, NONLOAD_WEIGHTS_RW2_SUM), + _ => (&NONLOAD_WEIGHTS, 75), + } + } + /// Shadow instructions per hash (0 without a shadow). pub fn shadow_instrs_per_hash(&self) -> usize { self.shadow.map(|s| s.instrs_per_hash()).unwrap_or(0) @@ -592,6 +695,29 @@ impl LoadClass { /// "mx4": the v3 construction; a trailing "m" and "g" set the mixer multiplier and the growth rule on any /// load class, "w16m4g" for example). pub fn parse(s: &str) -> Option { + // class v6 lane 1: "+fold" (the index fold) and "+rw" / "+rw2" (the re-weight table) over + // any class, in any order, outermost of all + // D2 "layer 8 off": "+nowin" over any class, in any order with the other suffixes + if let Some(base) = s.strip_suffix("+nowin") { + return Some(LoadClass::parse(base)?.with_nowin()); + } + if let Some(base) = s.strip_suffix("+fold") { + return Some(LoadClass::parse(base)?.with_fold()); + } + if let Some(base) = s.strip_suffix("+rw2") { + return Some(LoadClass::parse(base)?.with_rw(2)); + } + if let Some(base) = s.strip_suffix("+rw") { + return Some(LoadClass::parse(base)?.with_rw(1)); + } + // "+reg64c": the 64-register window with the full-chain address mix (outermost, a research class) + if let Some(base) = s.strip_suffix("+reg64c") { + return Some(LoadClass::parse(base)?.with_reg64().with_reg64_chain()); + } + // "+reg64": the 64-register window over any class (the suffix is outermost, a research class) + if let Some(base) = s.strip_suffix("+reg64") { + return Some(LoadClass::parse(base)?.with_reg64()); + } // "+state": the state leaves of class v5 over any class (the suffix is outermost) if let Some(base) = s.strip_suffix("+state") { return Some(LoadClass::parse(base)?.with_state()); @@ -691,6 +817,7 @@ impl LoadClass { "w4" => [100, 0, 0], "w16" => [0, 100, 0], "w64" => [0, 0, 100], + "w32" => return Some(LoadClass::fixed8(slots)), m => { let v: Vec = m.split(',').map(|x| x.trim().parse::().ok()).collect::>>()?; if v.len() != 3 || v.iter().map(|&x| x as u32).sum::() != 100 { @@ -719,6 +846,22 @@ impl LoadClass { /// An era class is the base name with "-era" appended ("w4-era401998a5", "mx4-era..."). /// A hot class appends "hotk[a]" ("hot64k4", "scr4k32+hot64k4a"; measured and not adopted). pub fn name(&self) -> String { + // class v6 lane 1: "+fold" then "+rw" / "+rw2" are the outermost suffixes ("mx8+sh256x27+state+fold+rw") + if self.rw != 0 { + return format!("{}+rw{}", LoadClass { rw: 0, ..*self }.name(), if self.rw == 1 { "" } else { "2" }); + } + if self.fold { + return format!("{}+fold", LoadClass { fold: false, ..*self }.name()); + } + if self.nowin { + // D2 "layer 8 off": "+nowin" sits inside "+fold" and "+rw" and outside "+reg64" and "+state" + return format!("{}+nowin", LoadClass { nowin: false, ..*self }.name()); + } + if self.reg64 { + // "+reg64" / "+reg64c": the 64-register window is a suffix on any class, outermost + let base = LoadClass { reg64: false, reg64_chain: false, ..*self }.name(); + return if self.reg64_chain { format!("{base}+reg64c") } else { format!("{base}+reg64") }; + } if self.state { // "+state": class v5's leaves are a suffix on any class, outermost return format!("{}+state", LoadClass { state: false, ..*self }.name()); @@ -763,6 +906,7 @@ impl LoadClass { format!("scr{k}k{}", self.scratch_kb) } else { let base = match self.mix { + [0, 100, 0] if self.wide8 => "w32".to_string(), [100, 0, 0] => "w4".to_string(), [0, 100, 0] => "w16".to_string(), [0, 0, 100] => "w64".to_string(), @@ -791,15 +935,15 @@ impl LoadClass { for (i, &m) in self.mix.iter().enumerate() { acc += m as u64; if roll < acc { - return WIDTH_WORDS[i]; + return self.widths()[i]; } } - WIDTH_WORDS[2] + self.widths()[2] } /// Expected dataset bytes read per hash: dataset loads per hash times the mean width (scratch traffic apart). pub fn expected_bytes_per_hash(&self) -> f64 { - let mean = self.mix.iter().zip(WIDTH_WORDS.iter()).map(|(&m, &w)| m as f64 / 100.0 * w as f64 * 4.0).sum::(); + let mean = self.mix.iter().zip(self.widths().iter()).map(|(&m, &w)| m as f64 / 100.0 * w as f64 * 4.0).sum::(); (self.load_slots as usize - self.scratch_slots()) as f64 * ITERATIONS as f64 * mean } } @@ -819,6 +963,10 @@ pub const GENERATOR_VERSION_V4: u32 = 4; /// Generator version of a class v5 program (proof of stored state and of following, 7 October 2026, PROPOSED: /// `program_id(5, seed, attempt)`; `docs/design/class-v5-stored-state.md`). pub const GENERATOR_VERSION_V5: u32 = 5; +/// Class v6 (the Igneum 2.0 D1 object, 8 October 2026; review B's F03): the class v5 draw with the v6 rules (the index fold, the +/// re-weight table, the 64-register window with the full chain, the five shuffle dimensions and the mad operand rule by +/// construction, layer 8 off when the flag says so), generator 6. +pub const GENERATOR_VERSION_V6: u32 = 6; /// The program class of an epoch (Counter ASIC 2.0, 5 October 2026, `docs/plans/counter-asic-2-rollout.md`): one /// height switch in the node, `program_class_v3_activation_daa`, rounded up to an epoch boundary, decides which @@ -835,6 +983,8 @@ pub enum ProgramClass { /// Class v5 (`docs/design/class-v5-stored-state.md`, behind `program_class_v5_activation_daa`): class v4's program /// over a dataset whose every item is keyed by the window's execution state ([`V5_CLASS`]), generator 5. V5, + /// Class v6 (the Igneum 2.0 D1 object; review B's F03, 8 October 2026): [`V6_CLASS`], generator 6. + V6, } /// The load class of program class v3, decided 5 October 2026 (Counter ASIC 2.0, `docs/plans/counter-asic-2-status.md` @@ -857,6 +1007,9 @@ pub const V4_CLASS: LoadClass = LoadClass { shadow: Some(ShadowClass { instrs: V /// program draw, the shadow block, the era draw and the ladder rung are class v4's, draw for draw; only the item /// derivation and the program id change. pub const V5_CLASS: LoadClass = LoadClass { state: true, ..V4_CLASS }; +/// The class v6 object ("mx8+sh256x27+state+reg64c+fold+rw"): class v5 with the index fold, the k lane's re-weight table +/// and the 64-register window with the full chain; layer 8 off is the fifth flag, drawn by the chain's family flags. +pub const V6_CLASS: LoadClass = LoadClass { fold: true, rw: 1, reg64: true, reg64_chain: true, ..V5_CLASS }; /// The shadow block size of class v4 at every rung of the latency ladder (`docs/design/latency-ladder.md`): 256 /// instructions. The ladder moves the pass count alone. @@ -921,6 +1074,8 @@ pub fn era_generator_of(base: &LoadClass) -> u32 { match ProgramClass::of_load_class(base) { Some(ProgramClass::V4) => GENERATOR_VERSION_V4, Some(ProgramClass::V5) => GENERATOR_VERSION_V5, + Some(ProgramClass::V6) => GENERATOR_VERSION_V6, + _ if base.is_v6() => GENERATOR_VERSION_V6, _ if base.state => GENERATOR_VERSION_V5, _ => GENERATOR_VERSION_V3, } @@ -951,6 +1106,7 @@ impl ProgramClass { ProgramClass::V3 => V3_CLASS, ProgramClass::V4 => V4_CLASS, ProgramClass::V5 => V5_CLASS, + ProgramClass::V6 => V6_CLASS, } } @@ -961,12 +1117,13 @@ impl ProgramClass { ProgramClass::V3 => GENERATOR_VERSION_V3, ProgramClass::V4 => GENERATOR_VERSION_V4, ProgramClass::V5 => GENERATOR_VERSION_V5, + ProgramClass::V6 => GENERATOR_VERSION_V6, } } /// Whether the class's dataset is keyed by the window's execution state (class v5). pub fn has_state(&self) -> bool { - *self == ProgramClass::V5 + matches!(self, ProgramClass::V5 | ProgramClass::V6) } /// The class of a generator version: 2, 3 and 4 are the three classes, anything else is no class this crate runs. @@ -976,6 +1133,7 @@ impl ProgramClass { GENERATOR_VERSION_V3 => Some(ProgramClass::V3), GENERATOR_VERSION_V4 => Some(ProgramClass::V4), GENERATOR_VERSION_V5 => Some(ProgramClass::V5), + GENERATOR_VERSION_V6 => Some(ProgramClass::V6), _ => None, } } @@ -987,6 +1145,7 @@ impl ProgramClass { ProgramClass::V3 => "v3", ProgramClass::V4 => "v4", ProgramClass::V5 => "v5", + ProgramClass::V6 => "v6", } } @@ -996,6 +1155,7 @@ impl ProgramClass { "v3" => Some(ProgramClass::V3), "v4" => Some(ProgramClass::V4), "v5" => Some(ProgramClass::V5), + "v6" => Some(ProgramClass::V6), _ => None, } } @@ -1008,7 +1168,8 @@ impl ProgramClass { /// The program class whose load class `class` is, the era draw set aside: [`LoadClass::V2`] is v2, [`V3_CLASS`] /// is v3, [`V4_CLASS`] is v4; a measurement class (a width, a derivation length, another shadow size) is none. pub fn of_load_class(class: &LoadClass) -> Option { - let base = LoadClass { era: None, ..*class }; + let base = LoadClass { era: None, reg64: false, reg64_chain: false, ..*class }; + let v6base = LoadClass { era: None, nowin: false, ..*class }; if base == LoadClass::V2 { Some(ProgramClass::V2) } else if base == V3_CLASS { @@ -1017,6 +1178,8 @@ impl ProgramClass { Some(ProgramClass::V4) } else if base == V5_CLASS { Some(ProgramClass::V5) + } else if v6base == V6_CLASS { + Some(ProgramClass::V6) } else { None } @@ -1032,10 +1195,71 @@ impl ProgramClass { } } +/// Registers per lane of a program: 64 under the reg64 flag, 8 for every other class. +pub const REG64_REGISTERS: usize = 64; +/// The reg64 full-chain fold in the program id: 1 = the load's own source out of the rotate-xor chain. +pub const REG64_CHAIN_FOLD: u8 = 1; + impl Program { + /// Registers per lane (8, or 64 under `class.reg64`). + pub fn registers(&self) -> usize { + if self.class.reg64 { REG64_REGISTERS } else { 8 } + } + /// Whether every load's address consumes all 64 registers (the reg64 full-chain variant). + pub fn address_mix(&self) -> bool { + self.class.reg64 && self.class.reg64_chain + } + /// The pack's liveness statement (the reg64 variants): which reads keep every register necessary. + pub fn reg64_liveness(&self) -> String { + if !self.class.reg64 { + return String::new(); + } + let loads = self.scheduled().iter().filter(|i| i.op == Op::Load).count(); + if self.address_mix() { + format!("every one of the 64 registers is read by the address of each of the {loads} loads per iteration (the load's source directly, the 63 others through the rotate-xor chain) and by the end fold; 64 independently necessary values for the length of the dependent chain; xoring any register with either of two seed-derived probe words at the start of an iteration moves the first load address and the final hash") + } else { + "every one of the 64 registers is read by the end fold (r[k] ^= r[k + 8] ^ ... ^ r[k + 56]); a load's address reads its own window's register only, so the chain needs 8 live values at a time and the other 56 are live across it as state (window, arithmetic-only)".to_string() + } + } + /// The instructions one iteration executes, in order, with their register fields as the kernels name them. + /// Without the reg64 flag this is the drawn program as it stands. Under the flag, instruction i of the drawn + /// program runs on window 0 with every register field widened by `8 * (i % 4)` (so the 64 instructions touch all + /// 32 registers of the window), then the same instruction runs on window 1 (the widened field + 32): 128 + /// statements per iteration over registers 0..63. The CPU verifier and every emitter read this list, so the + /// vectors and the kernel text agree by construction. + pub fn scheduled(&self) -> Vec { + if !self.class.reg64 { + return self.instrs.clone(); + } + let mut out = Vec::with_capacity(self.instrs.len() * 2); + for (i, ins) in self.instrs.iter().enumerate() { + let off = (8 * (i % 4)) as u8; + let a = Instr { dst: ins.dst + off, src: ins.src + off, src2: ins.src2 + off, ..*ins }; + let b = Instr { dst: a.dst + 32, src: a.src + 32, src2: a.src2 + 32, ..a }; + out.push(a); + out.push(b); + } + out + } + /// The drawn program's memory operations per nonce (16 load slots times the eight iterations on every class): + /// the count the acceptance rule and the draw's bookkeeping read (the acceptance judges the drawn program's + /// sites). The hash EXECUTES [`Program::loads_per_hash_executed`] of them, twice this under the 64-register + /// window; the served figures and the pack's headers name the executed count (review B's V6-02). pub fn loads_per_hash(&self) -> usize { self.instrs.iter().filter(|i| i.op.is_load()).count() * ITERATIONS } + + /// Memory operations the hash EXECUTES per nonce (review B's V6-02, 8 October 2026): the scheduled statements + /// (the drawn program's loads, twice under the 64-register window) times the eight iterations; asserted to + /// agree with the drawn count except by the window's factor of two. The emitters alone read it (program.h, + /// program.json): the acceptance reads the drawn count, so no verdict moves with the served figure (the first + /// form of this change, 0266c9ec0, routed the executed count into the acceptance and moved every reg64 verdict). + pub fn loads_per_hash_executed(&self) -> usize { + let drawn = self.loads_per_hash(); + let executed = self.scheduled().iter().filter(|i| i.op.is_load()).count() * ITERATIONS; + assert!(executed == drawn || (self.class.reg64 && executed == 2 * drawn), "the executed load count {executed} disagrees with the drawn {drawn}"); + executed + } pub fn wide_loads_per_hash(&self) -> usize { self.instrs.iter().filter(|i| i.op == Op::WLoad).count() * ITERATIONS } @@ -1043,9 +1267,21 @@ impl Program { self.instrs.iter().any(|i| i.op == Op::WLoad) } /// Dataset bytes read per hash: 4 per one-word load, 16 and 64 for the wider loads of the experiment. + /// Dataset bytes the drawn program's loads demand per nonce (the acceptance's and the draw's figure). pub fn bytes_per_hash(&self) -> usize { self.instrs.iter().filter(|i| i.op == Op::Load).map(|i| i.width as usize * 4).sum::() * ITERATIONS } + + /// Dataset bytes the hash's EXECUTED loads demand per nonce (review B's V6-02): the scheduled loads' widths in + /// words times 4, times the eight iterations; a demand figure, not the memory's transactions (a 4-byte load is + /// a 32-byte sector on NVIDIA and a 64-byte line on AMD; the served text names the model it quotes). The + /// emitters alone read it. + pub fn bytes_per_hash_executed(&self) -> usize { + let drawn = self.bytes_per_hash(); + let executed = self.scheduled().iter().filter(|i| i.op == Op::Load).map(|i| i.width as usize * 4).sum::() * ITERATIONS; + assert!(executed == drawn || (self.class.reg64 && executed == 2 * drawn), "the executed byte demand {executed} disagrees with the drawn {drawn}"); + executed + } /// Scratch read-modify-writes per hash (variant 5): each reads 16 bytes and writes 16 bytes. pub fn scratch_ops_per_hash(&self) -> usize { self.instrs.iter().filter(|i| i.op == Op::Scratch).count() * ITERATIONS @@ -1093,7 +1329,7 @@ impl Program { pub fn width_counts(&self) -> [usize; 3] { let mut c = [0usize; 3]; for i in self.instrs.iter().filter(|i| i.op == Op::Load) { - if let Some(k) = WIDTH_WORDS.iter().position(|&w| w == i.width) { + if let Some(k) = self.class.widths().iter().position(|&w| w == i.width) { c[k] += 1; } } @@ -1137,7 +1373,7 @@ impl Program { let v4_rung_0 = self.generator == GENERATOR_VERSION_V4 && LoadClass { era: None, ..self.class } == V4_CLASS; // class v5 at rung 0 is `program_id(5, seed, attempt)`; above rung 0 the class-bearing id with "state/" let v5_rung_0 = self.generator == GENERATOR_VERSION_V5 && LoadClass { era: None, ..self.class } == V5_CLASS; - if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 || v5_rung_0 { + if !self.class.reg64 && (self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 || v5_rung_0) { // Spec 01 section 1.4.6: a class v3 program's id is `program_id(3, seed, attempt)`, a class v4 program's // `program_id(4, seed, attempt)` (Counter ASIC 3.0); the generator version in the preimage separates // them from every version 2 program of the same seed @@ -1147,35 +1383,87 @@ impl Program { } } + /// The derivation of [`Program::program_id`] as text, from the same byte recipe (program.json's + /// "program_id_derivation"; spec 01 section 1.4.6). + pub fn program_id_derivation(&self) -> String { + let v4_rung_0 = self.generator == GENERATOR_VERSION_V4 && LoadClass { era: None, ..self.class } == V4_CLASS; + // class v5 at rung 0 takes the plain form as `program_id` does (the text follows the id; the first form of this + // function named the class recipe for the v5 packs whose id was the plain one) + let v5_rung_0 = self.generator == GENERATOR_VERSION_V5 && LoadClass { era: None, ..self.class } == V5_CLASS; + if !self.class.reg64 && (self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 || v5_rung_0) { + program_id_recipe(self.generator, &self.seed, self.attempt).text() + } else { + program_id_class_recipe(self.generator, &self.seed, self.attempt, &self.class).text() + } + } + /// The program class of this program, from its generator version (3 = v3, 4 = v4, everything else v2). pub fn program_class(&self) -> ProgramClass { match self.generator { GENERATOR_VERSION_V3 => ProgramClass::V3, GENERATOR_VERSION_V4 => ProgramClass::V4, GENERATOR_VERSION_V5 => ProgramClass::V5, + GENERATOR_VERSION_V6 => ProgramClass::V6, _ => ProgramClass::V2, } } } -pub fn program_id(generator: u32, seed: &[u32; 8], attempt: u32) -> u64 { - let mut b = Vec::with_capacity(PROGRAM_ID_TAG.len() + 4 + 32 + 4 + 6); - b.extend_from_slice(PROGRAM_ID_TAG); - b.extend_from_slice(&generator.to_le_bytes()); - for w in seed { - b.extend_from_slice(&w.to_le_bytes()); +/// The byte recipe of a program id and the text that states it: the bytes FNV-1a 64 hashes and, part by part, the label +/// of each part (`'literal'` or `field_le32`), so the derivation printed in program.json is built from the same list the id +/// is hashed from and the two cannot drift (the documentation finding of 7 October 2026: program.json and spec 1.4.6 said +/// the plain form while generator 4 appended `'sub/' || sub_version_le16`, so a client written from the text derived another +/// id). `tests/derivation.rs` re-derives every pinned pack's id from its own derivation string. +#[derive(Clone, Debug, Default)] +pub struct IdRecipe { + pub bytes: Vec, + parts: Vec, +} + +impl IdRecipe { + fn lit(&mut self, s: &[u8]) { + self.bytes.extend_from_slice(s); + self.parts.push(format!("'{}'", String::from_utf8_lossy(s))); } - b.extend_from_slice(&attempt.to_le_bytes()); + fn field(&mut self, bytes: &[u8], label: &str) { + self.bytes.extend_from_slice(bytes); + self.parts.push(label.to_string()); + } + /// The derivation as program.json prints it: `FNV-1a 64 over || || ...`. + pub fn text(&self) -> String { + format!("FNV-1a 64 over {}", self.parts.join(" || ")) + } + /// The id: FNV-1a 64 over the bytes. + pub fn id(&self) -> u64 { + fnv1a64(&self.bytes) + } +} + +/// The recipe of [`program_id`]. +pub fn program_id_recipe(generator: u32, seed: &[u32; 8], attempt: u32) -> IdRecipe { + let mut r = IdRecipe::default(); + r.lit(PROGRAM_ID_TAG); + r.field(&generator.to_le_bytes(), "generator_le32"); + let mut sw = Vec::with_capacity(32); + for w in seed { + sw.extend_from_slice(&w.to_le_bytes()); + } + r.field(&sw, "seed_words as little-endian bytes"); + r.field(&attempt.to_le_bytes(), "attempt_le32"); if generator == GENERATOR_VERSION_V4 { // The class v4 sub-version (AP-F8-1 amendment, 7 October 2026): `"sub/" || sub_version as little-endian u16` // appended for generator 4 only, so a binary from before the load-source rule (sub-version 0, no suffix) and // one after it never share a program id for one seed; the node's id check then catches a split. v2 and v3 // ids are byte-identical. The node reads the sub-version from [`PROGRAM_SUBVERSION_V4`]; packs carry it as // IGNEUM_PROGRAM_SUBVERSION and program.json "sub_version". - b.extend_from_slice(b"sub/"); - b.extend_from_slice(&PROGRAM_SUBVERSION_V4.to_le_bytes()); + r.lit(b"sub/"); + r.field(&PROGRAM_SUBVERSION_V4.to_le_bytes(), "sub_version_le16"); } - fnv1a64(&b) + r +} + +pub fn program_id(generator: u32, seed: &[u32; 8], attempt: u32) -> u64 { + program_id_recipe(generator, seed, attempt).id() } /// The sub-version of class v4's program stream, in every generator-4 program id and pack (AP-F8-3: 3 = the @@ -1189,57 +1477,92 @@ pub const PROGRAM_ID_TAG_RW: &[u8] = b"igneum-program-rw/"; /// The program id of a non-default class: the tag, then the same fields as [`program_id`], then the three mix /// percentages and the slot count as bytes. -pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &LoadClass) -> u64 { - let mut b = Vec::with_capacity(PROGRAM_ID_TAG_RW.len() + 4 + 32 + 4 + 4); - b.extend_from_slice(PROGRAM_ID_TAG_RW); - b.extend_from_slice(&generator.to_le_bytes()); +/// The recipe of [`program_id_class`]. +pub fn program_id_class_recipe(generator: u32, seed: &[u32; 8], attempt: u32, class: &LoadClass) -> IdRecipe { + let mut r = IdRecipe::default(); + r.lit(PROGRAM_ID_TAG_RW); + r.field(&generator.to_le_bytes(), "generator_le32"); + let mut sw = Vec::with_capacity(32); for w in seed { - b.extend_from_slice(&w.to_le_bytes()); + sw.extend_from_slice(&w.to_le_bytes()); } - b.extend_from_slice(&attempt.to_le_bytes()); - b.extend_from_slice(&class.mix); - b.push(class.load_slots); + r.field(&sw, "seed_words as little-endian bytes"); + r.field(&attempt.to_le_bytes(), "attempt_le32"); + r.field(&class.mix, "mix[3]"); + r.field(&[class.load_slots], "load_slots_u8"); if let Some(k) = class.scratch { - b.extend_from_slice(b"scratch/"); - b.push(k); - b.push(class.scratch_kb); + r.lit(b"scratch/"); + r.field(&[k], "scratch_k_u8"); + r.field(&[class.scratch_kb], "scratch_kb_u8"); } if class.mixer_mult != 1 || class.growth { // Counter ASIC 2.0: the mixer multiplier and the growth rule are part of the construction, so a program of // the same seed under a different mixer carries a different id (under the v3 seam the id is // program_id(3, seed, attempt) and this branch is not taken) - b.extend_from_slice(b"mixer/"); - b.push(class.mixer_mult); - b.push(class.growth as u8); + r.lit(b"mixer/"); + r.field(&[class.mixer_mult], "mixer_mult_u8"); + r.field(&[class.growth as u8], "growth_u8"); } if class.derive_len != 0 { // Counter ASIC 3.0 item 2: the derivation program's length is part of the construction - b.extend_from_slice(b"derive/"); - b.extend_from_slice(&class.derive_len.to_le_bytes()); + r.lit(b"derive/"); + r.field(&class.derive_len.to_le_bytes(), "derive_len_le16"); } if let Some(e) = class.era { - b.extend_from_slice(b"era/"); - b.extend_from_slice(&e.id_bytes()); + r.lit(b"era/"); + r.field(&e.id_bytes(), "era_id_bytes (allowed[3] || width_words_u8 || stride_mul_le32 || stride_rot_le32 || interleave[4])"); } if let Some(sh) = class.shadow { // Counter ASIC 3.0 item 8: the shadow block's size and repeat count are part of the construction - b.extend_from_slice(b"shadow/"); - b.extend_from_slice(&sh.instrs.to_le_bytes()); - b.extend_from_slice(&sh.reps.to_le_bytes()); + r.lit(b"shadow/"); + r.field(&sh.instrs.to_le_bytes(), "shadow_instrs_le16"); + r.field(&sh.reps.to_le_bytes(), "shadow_reps_le16"); } if let Some(h) = class.hot { - b.extend_from_slice(b"hot/"); - b.push(h.mb); - b.push(h.k); + r.lit(b"hot/"); + r.field(&[h.mb], "hot_mb_u8"); + r.field(&[h.k], "hot_k_u8"); if h.added { - b.extend_from_slice(b"added"); + r.lit(b"added"); } } if class.state { // class v5: the state leaves are part of the construction - b.extend_from_slice(b"state/"); + r.lit(b"state/"); } - fnv1a64(&b) + if class.fold { + // class v6 lane 1: the index fold moves every era load address + r.lit(b"fold/"); + } + if class.nowin { + // D2 "layer 8 off": the window layer is part of the construction + r.lit(b"nowin/"); + } + if class.is_v6() { + // the v6 draw rules (A02 five shuffle dimensions, A08 the mad operand rule) are part of the construction + r.lit(b"v6draw/"); + } + if class.rw != 0 { + // class v6 lane 1: the re-weight table moves the op draw + r.lit(b"rw/"); + r.field(&[class.rw], "rw_table_u8"); + } + if class.reg64 { + // the 64-register window is part of the construction + r.lit(b"reg64/"); + if class.reg64_chain { + // the fold text is part of the construction: 1 = the load's own source out of the chain (the sound + // fold); the benched text of 8 October 2026 (the source inside the chain, id 0x3deee2320e70e1bf for the + // pinned devnet seeds) carried no byte here, so the two texts never share an id + r.lit(b"chain/"); + r.field(&[REG64_CHAIN_FOLD], "reg64_chain_fold_u8"); + } + } + r +} + +pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &LoadClass) -> u64 { + program_id_class_recipe(generator, seed, attempt, class).id() } /// Weights of the ten non-load families under version 2, in draw order. Sum 75. The load family has no @@ -1257,6 +1580,38 @@ pub const NONLOAD_WEIGHTS: [(Op, u64); 10] = [ (Op::Or, 4), ]; +/// Class v6 lane 1, the re-weight table behind the index fold (`+rw`): the k lane's optimiser split in draw order, +/// sum [`NONLOAD_WEIGHTS_RW_SUM`] (83). Shuffle stays at 4 (a shuffle-heavy draw is the worst thing the class can do +/// on the chip side, k 0.021 routed) and `or` is never drawn. The roll ranges over the table's own sum. +pub const NONLOAD_WEIGHTS_RW: [(Op, u64); 10] = [ + (Op::Add, 16), + (Op::Xor, 14), + (Op::Mul, 4), + (Op::Mad, 12), + (Op::Shfl, 4), + (Op::Rotl, 11), + (Op::Sub, 10), + (Op::MulHi, 2), + (Op::Rotr, 10), + (Op::Or, 0), +]; +pub const NONLOAD_WEIGHTS_RW_SUM: u64 = 83; + +/// Class v6 lane 1, the census lane's neighbouring table (`+rw2`), sum [`NONLOAD_WEIGHTS_RW2_SUM`] (75). +pub const NONLOAD_WEIGHTS_RW2: [(Op, u64); 10] = [ + (Op::Add, 13), + (Op::Xor, 11), + (Op::Mul, 6), + (Op::Mad, 10), + (Op::Shfl, 8), + (Op::Rotl, 8), + (Op::Sub, 7), + (Op::MulHi, 2), + (Op::Rotr, 6), + (Op::Or, 4), +]; +pub const NONLOAD_WEIGHTS_RW2_SUM: u64 = 75; + /// Version 1 weights (retired). Sum 100, load at 25 percent. pub const OP_WEIGHTS: [(Op, u64); 11] = [ (Op::Load, 25), @@ -1355,6 +1710,23 @@ pub fn candidate_from_words_class( for &slot in &p[..slots] { is_load[slot as usize] = true; } + // class v6 (A02, by construction): five of the non-load slots are shuffles over the five lane dimensions in a + // drawn order; drawn from the stream after the load slots, so no other class's stream moves + let mut is_shfl = [false; INSTR_COUNT]; + let mut shfl_masks = [1u8, 2, 4, 8, 16]; + if class.is_v6() { + let rest = &mut p[slots..]; + for i in 0..5 { + let j = i + rng.below((rest.len() - i) as u64) as usize; + rest.swap(i, j); + is_shfl[rest[i] as usize] = true; + } + for i in 0..5 { + let j = i + rng.below((5 - i) as u64) as usize; + shfl_masks.swap(i, j); + } + } + let mut shfl_next = 0usize; // Variant 5: the first k drawn load slots (a uniform k-subset, the draw order is random) are scratch ops. let mut is_scratch = [false; INSTR_COUNT]; for &slot in &p[..class.scratch_slots()] { @@ -1378,8 +1750,12 @@ pub fn candidate_from_words_class( // era set aside) on EVERY draw path, era or not, so a census through candidate_class reads the same stream as // the chain; v2, v3 and every other class take no part. The draw order and the stream are otherwise the same. // class v5 (docs/design/class-v5-stored-state.md) draws under the same rule: its state flag is set aside here too + // class v6 lane 1: the index fold and the re-weight table are set aside too (the address path and the table are not + // the shape; a +fold or +rw program draws its sources under the same rule) let source_rule_v4 = matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. })) - && LoadClass { era: None, shadow: None, state: false, ..class } == LoadClass { shadow: None, ..V4_CLASS }; + && LoadClass { era: None, shadow: None, state: false, fold: false, rw: 0, nowin: false, reg64: false, reg64_chain: false, mix: V4_CLASS.mix, ..class } == LoadClass { shadow: None, ..V4_CLASS }; + // the op table and the roll's range: the plain table at 75 for every class without the re-weight flag + let (weights, weights_sum) = class.nonload_weights(); let mut fresh = [false; 8]; let mut fresh_value = [true; 8]; // the shared-operand idiom (AP-F8-1, sub-version 3): after `or d |= s`, a later `xor d ^= s` or `sub d -= s` with @@ -1389,9 +1765,9 @@ pub fn candidate_from_words_class( let mut pair_op: [Option<(Op, usize)>; 8] = [None; 8]; let mut instrs = Vec::with_capacity(INSTR_COUNT); for k in 0..INSTR_COUNT { - let mut roll = rng.below(75); + let mut roll = rng.below(weights_sum); let mut op = Op::Add; - for &(o, w) in &NONLOAD_WEIGHTS { + for &(o, w) in weights { if roll < w { op = o; break; @@ -1407,6 +1783,9 @@ pub fn candidate_from_words_class( Op::Load }; } + if is_shfl[k] { + op = Op::Shfl; + } let dst = rng.below(8); let src = if op.is_load() { let mut eligible = [0u64; 8]; @@ -1435,12 +1814,21 @@ pub fn candidate_from_words_class( a } }; - let b = rng.below(8); + let mut b = rng.below(8); + // class v6 (A08, by construction): a mad's second source is never its destination + if class.is_v6() && op == Op::Mad && b == dst { + b = (b + 1) & 7; + } let imm = rng.next() as u32; let imm2 = rng.next() as u32; let rot = 1 + rng.below(31) as u32; let bit = rng.below(32); - let mask = 1u8 << rng.below(5); + let mut mask = 1u8 << rng.below(5); + // class v6 (A02): the reserved shuffle slots take the five dimensions in the drawn order + if is_shfl[k] { + mask = shfl_masks[shfl_next]; + shfl_next += 1; + } // Version 2 loads take no width roll, so a mixer class with version 2 loads draws the version 2 program let width = if class.takes_width_roll() { class.width_for_roll(rng.below(100)) } else { 1 }; let width = if op == Op::Load { width } else { 1 }; @@ -1448,7 +1836,8 @@ pub fn candidate_from_words_class( let (win, off) = if class.era.is_some() { let k = rng.below(3) as u8; let o = (rng.next() as u32 & ((1u32 << k) - 1)) as u8; - if op == Op::Load { + // D2 "layer 8 off": the draws are consumed (the stream is the class's) and every site reads the whole dataset + if op == Op::Load && !class.nowin { (k, o) } else { (0, 0) @@ -1496,9 +1885,9 @@ pub fn candidate_from_words_class( loop { shadow.clear(); for _ in 0..sh.instrs { - let mut roll = rng.below(75); + let mut roll = rng.below(weights_sum); let mut op = Op::Add; - for &(o, w) in &NONLOAD_WEIGHTS { + for &(o, w) in weights { if roll < w { op = o; break; @@ -1589,7 +1978,12 @@ pub fn try_generate_class(seed_string: &str, seed_bytes: &[u8], class: LoadClass } if crate::accept::is_class_v4_shape(&class) { // AP-F8-2 (7 October 2026, main's ruling: the draw is total and no consensus path panics): a class v4 seed that - // exhausts its attempts takes the last-resort program, deterministic and accepted as drawn + // exhausts its attempts takes the last-resort program, deterministic. Sub-version 3's is accepted as drawn + // (unreachable at 4.6e-44 per epoch and unverified against the rule: adv-accept-3's finding, 223 of 2,500 + // rewritten candidates fail it, 209 by part (a)); class v5's is the verified one. + if class.state { + return Ok(last_resort_v5(seed_string, seed_bytes, cap, class)); + } return Ok(last_resort_v4(candidate_class(seed_string, seed_bytes, cap, class))); } Err(Exhausted { seed_string: seed_string.to_string(), attempts: cap, last: last.unwrap() }) @@ -1613,6 +2007,55 @@ pub fn last_resort_v4(mut p: Program) -> Program { p } +/// Class v5's last resort, verified (adv-accept-3's finding of 7 October 2026: the sub-version 3 rewrite alone fails the +/// rule on 223 of 2,500 seeds, 209 by part (a), a load reading a register no instruction wrote since the previous load +/// from it). From attempt `cap` on, each candidate is rewritten as [`last_resort_v4`] does, its stale loads re-sourced +/// by [`repair_stale_loads`], and the result checked against the whole rule; the first that passes is the program. The +/// scan runs [`LAST_RESORT_SCAN`] candidates; past it the first repaired candidate stands as drawn, so the draw is total, +/// and that fallback sits behind the 4.6e-44 of reaching the last resort at all times the rejection rate of a repaired +/// candidate (about 0.09 per adv-accept-3) to the power of the scan: under 1e-300. +pub fn last_resort_v5(seed_string: &str, seed_bytes: &[u8], cap: u32, class: LoadClass) -> Program { + for k in cap..cap + LAST_RESORT_SCAN { + let p = repair_stale_loads(last_resort_v4(candidate_class(seed_string, seed_bytes, k, class))); + if check(&p).is_ok() { + return p; + } + } + repair_stale_loads(last_resort_v4(candidate_class(seed_string, seed_bytes, cap, class))) +} + +/// The candidates class v5's last resort scans past the attempt cap before the unchecked fallback. +pub const LAST_RESORT_SCAN: u32 = 256; + +/// Part (a) by construction: a load whose source register no instruction wrote since the previous load from it (in +/// cyclic order over the 64 instructions, the two-pass walk of the acceptance's `check_stale_loads`) is re-sourced to +/// the lowest register that was written since its last load. The walk repeats until a full two-pass walk changes +/// nothing (a re-sourced load can make a later load from the new register stale, which the next walk re-sources); +/// with 16 loads among 64 instructions a non-pending register always exists. Only the `src` field moves. +pub fn repair_stale_loads(mut p: Program) -> Program { + for _round in 0..16 { + let mut changed = false; + let mut pending = [false; 8]; + for _pass in 0..2 { + for ins in p.instrs.iter_mut() { + if ins.op.is_load() && pending[ins.src as usize] { + let q = (0..8usize).find(|&q| !pending[q]).expect("a register written since its last load") as u8; + ins.src = q; + changed = true; + } + pending[ins.dst as usize] = false; + if ins.op.is_load() { + pending[ins.src as usize] = true; + } + } + } + if !changed { + break; + } + } + p +} + /// [`try_generate_from_seed_bytes`], treating exhaustion as the consensus fault it is. pub fn generate_from_seed_bytes(seed_string: &str, seed_bytes: &[u8]) -> Program { try_generate_from_seed_bytes(seed_string, seed_bytes).unwrap_or_else(|e| panic!("{e}")) @@ -1634,6 +2077,7 @@ pub fn generate_from_seed_bytes_program_class(seed_string: &str, seed_bytes: &[u (ProgramClass::V3, Some(era)) => return generate_era(seed_string, seed_bytes, V3_CLASS, era, &V3_ALLOWED), (ProgramClass::V4, Some(era)) => return generate_era_generator(seed_string, seed_bytes, V4_CLASS, era, &V3_ALLOWED, GENERATOR_VERSION_V4), // Class v5: the same draw inside V5_CLASS (V4_CLASS plus the state flag, which the draw does not read), generator 5. + (ProgramClass::V6, Some(era)) => return generate_era_generator(seed_string, seed_bytes, V6_CLASS, era, &V3_ALLOWED, GENERATOR_VERSION_V6), (ProgramClass::V5, Some(era)) => return generate_era_generator(seed_string, seed_bytes, V5_CLASS, era, &V3_ALLOWED, GENERATOR_VERSION_V5), _ => {} } @@ -1651,11 +2095,11 @@ pub fn generate_from_seed_bytes_program_class(seed_string: &str, seed_bytes: &[u /// and class v4 at rung 0, is [`generate_from_seed_bytes_program_class`] byte for byte. The base program, the 16 loads /// and the era draw do not move with the rung: only the pass count of the shadow block does. pub fn generate_from_seed_bytes_program_class_shadow(seed_string: &str, seed_bytes: &[u8], class: ProgramClass, era_bytes: Option<&[u8]>, shadow_reps: u16) -> Program { - if !matches!(class, ProgramClass::V4 | ProgramClass::V5) || shadow_reps == 0 || shadow_reps == V4_SHADOW_REPS { + if !matches!(class, ProgramClass::V4 | ProgramClass::V5 | ProgramClass::V6) || shadow_reps == 0 || shadow_reps == V4_SHADOW_REPS { return generate_from_seed_bytes_program_class(seed_string, seed_bytes, class, era_bytes); } // class v5 at a rung: class v4's rung with the state flag, generator 5 - let (base, generator) = if class == ProgramClass::V5 { (v5_class_at(shadow_reps), GENERATOR_VERSION_V5) } else { (v4_class_at(shadow_reps), GENERATOR_VERSION_V4) }; + let (base, generator) = if class == ProgramClass::V6 { (LoadClass { fold: true, rw: 1, reg64: true, reg64_chain: true, ..v5_class_at(shadow_reps) }, GENERATOR_VERSION_V6) } else if class == ProgramClass::V5 { (v5_class_at(shadow_reps), GENERATOR_VERSION_V5) } else { (v4_class_at(shadow_reps), GENERATOR_VERSION_V4) }; match era_bytes { Some(era) => generate_era_generator(seed_string, seed_bytes, base, era, &V3_ALLOWED, generator), None => { @@ -1898,6 +2342,14 @@ mod tests { assert!(v2.instrs.iter().all(|i| i.width == 1)); assert_eq!(v2.bytes_per_hash(), 512); assert_eq!(LoadClass::parse("w16"), Some(LoadClass::fixed(4, 16))); + // W = 8 (8 October 2026): the w32 class reads 8 words a load; w16 reads 4 (known-failed first: the plain set has no 8) + assert!(!WIDTH_WORDS.contains(&8)); + let w32 = LoadClass::parse("w32").unwrap(); + assert!(w32.wide8 && w32.mix == [0, 100, 0] && w32.widths() == [1, 8, 16]); + assert_eq!(w32.name(), "w32"); + assert_eq!(LoadClass::parse("w32x8").unwrap().load_slots, 8); + assert_eq!(w32.expected_bytes_per_hash(), 2.0 * LoadClass::fixed(4, 16).expected_bytes_per_hash()); + assert!(!LoadClass::fixed(4, 16).wide8); assert_eq!(LoadClass::parse("w64x4"), Some(LoadClass::fixed(16, 4))); assert_eq!(LoadClass::parse("50,35,15"), Some(LoadClass::mixed([50, 35, 15]))); assert_eq!(LoadClass::parse("v2"), Some(LoadClass::V2)); @@ -1993,14 +2445,15 @@ mod tests { assert_eq!(v3.program_id(), program_id(GENERATOR_VERSION_V3, &v3.seed, v3.attempt)); assert_ne!(v3.program_id(), program_id(GENERATOR_VERSION, &v3.seed, v3.attempt)); assert_ne!(v3.program_id(), v2.program_id()); - for c in [ProgramClass::V2, ProgramClass::V3, ProgramClass::V4, ProgramClass::V5] { + for c in [ProgramClass::V2, ProgramClass::V3, ProgramClass::V4, ProgramClass::V5, ProgramClass::V6] { assert_eq!(ProgramClass::parse(c.name()), Some(c)); assert_eq!(ProgramClass::from_generator(c.generator_version()), Some(c)); assert_eq!(ProgramClass::from_u8(c.as_u8()), Some(c)); } assert_eq!(ProgramClass::from_generator(1), None); - assert_eq!(ProgramClass::from_generator(6), None); - assert_eq!(ProgramClass::parse("v6"), None); + assert_eq!(ProgramClass::from_generator(6), Some(ProgramClass::V6), "class v6 is generator 6 (review B's F03, 8 October 2026)"); + assert_eq!(ProgramClass::from_generator(7), None); + assert_eq!(ProgramClass::parse("v7"), None); assert_eq!(ProgramClass::default(), ProgramClass::V2); assert_eq!(ProgramClass::V2.load_class(), LoadClass::V2); assert!(!ProgramClass::V2.has_era() && ProgramClass::V3.has_era() && ProgramClass::V4.has_era() && ProgramClass::V5.has_era()); @@ -2273,6 +2726,44 @@ mod tests { } } + /// Class v5's last resort is verified (adv-accept-3's finding, 7 October 2026): on seed adv3/steer/2 of that lane's + /// label space (`seed_words_from_bytes("igneum-adv-accept-3/steer/2")` as the epoch bytes, Devnet 3's genesis as the + /// era) the sub-version 3 rewrite of the candidate at the cap fails the rule by part (a), a cyclic stale load, and + /// would be handed to the chain; class v5's repair re-sources the load and the result passes the whole rule, as + /// does the program `last_resort_v5` returns. The class v4 path is unchanged (frozen, unreachable, recorded). + #[test] + fn class_v5_last_resort_is_verified_known_failed_adv3_steer_2() { + use crate::accept::{check, check_static, Reject}; + use crate::seed::seed_words_from_bytes; + let epoch: Vec = seed_words_from_bytes(b"igneum-adv-accept-3/steer/2").iter().flat_map(|x| x.to_le_bytes()).collect(); + let era = crate::bind::unhex("4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925").unwrap(); + let v4 = LoadClass::era(V4_CLASS, &era, &V3_ALLOWED); + let v5 = LoadClass::era(V5_CLASS, &era, &V3_ALLOWED); + let cap = max_attempts_for(&v4); + let old = last_resort_v4(candidate_class("adv3/steer/2", &epoch, cap, v4)); + match check_static(&old) { + Err(Reject::StaleLoadSource { instr, reg }) => println!("adv3/steer/2: the sub-version 3 last resort fails part (a) at instruction {instr} reading r{reg}"), + other => panic!("the known-failed case must fail part (a): {other:?}"), + } + let repaired = repair_stale_loads(old.clone()); + assert!(check_static(&repaired).is_ok(), "the repair restores part (a): {:?}", check_static(&repaired).err()); + assert_eq!(repaired.instrs.len(), old.instrs.len()); + assert!(repaired.instrs.iter().zip(old.instrs.iter()).all(|(a, b)| a.op == b.op && a.dst == b.dst), "only load sources move"); + let lr = last_resort_v5("adv3/steer/2", &epoch, cap, v5); + assert!(check(&lr).is_ok(), "class v5's last resort passes the whole rule: {:?}", check(&lr).err()); + assert!(lr.attempt >= cap && lr.attempt < cap + LAST_RESORT_SCAN); + assert!(lr.class.state); + println!("adv3/steer/2: class v5 last resort at attempt {} id {:016x}", lr.attempt, lr.program_id()); + // the sub-version 3 path is byte for byte what it was + assert_eq!(try_generate_class("adv3/steer/2", &epoch, v4).map(|p| p.attempt).ok(), Some(try_generate_class("adv3/steer/2", &epoch, v4).unwrap().attempt)); + // a few more of the label space: every class v5 last resort passes + for i in [11u32, 33, 56, 58, 77] { + let e: Vec = seed_words_from_bytes(format!("igneum-adv-accept-3/steer/{i}").as_bytes()).iter().flat_map(|x| x.to_le_bytes()).collect(); + let lr = last_resort_v5(&format!("adv3/steer/{i}"), &e, cap, v5); + assert!(check(&lr).is_ok(), "steer/{i}: {:?}", check(&lr).err()); + } + } + #[test] fn class_v4_draw_is_total_with_the_last_resort() { assert_eq!(max_attempts_for(&V4_CLASS), MAX_ATTEMPTS_V4); diff --git a/igneum-pow/src/main.rs b/igneum-pow/src/main.rs index 95795cd5d..205f6ce66 100644 --- a/igneum-pow/src/main.rs +++ b/igneum-pow/src/main.rs @@ -1,7 +1,7 @@ //! igneum-pow CLI. //! //! igneum-pow bench --seed [--day ] [--closed-form] [--dataset-log2 28] [--warps 20] -//! igneum-pow export --seed --out [--day ] [--closed-form] [--dataset-log2 28] [--epoch-hex <64 hex> --day-hex ] +//! igneum-pow export --seed --out [--day ] [--closed-form] [--dataset-log2 28 | --dataset-words N] [--epoch-hex <64 hex> --day-hex ] //! igneum-pow hash --seed --nonce [--day ] [--closed-form] [--dataset-log2 28] //! igneum-pow hash-bound --seed --prehash <64 hex> --nonce [--day ] [--closed-form] [--dataset-log2 28] //! [--epoch-hex <64 hex> --day-hex ] byte seeds instead of strings (Epoch::from_seed_bytes) @@ -22,7 +22,7 @@ use igneum_pow::emit::export_pack; use igneum_pow::generator::{LoadClass, ProgramClass}; use igneum_pow::memhard::{Cache, Shape}; use igneum_pow::seed::day_key; -use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2}; +use igneum_pow::verify::{DatasetGeom, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2}; use std::time::Instant; struct Args { @@ -32,6 +32,10 @@ struct Args { out: Option, closed_form: bool, dataset_log2: u32, + /// `--dataset-words N`: the dataset at N words (research class ds55, 8 October 2026): any multiple of 2^16 in + /// 2^28 ..= 2^31; a non-power-of-two takes the multiply-shift range reduction of spec 01 section 1.13.3 in every + /// load, a power of two is the `--dataset-log2` path. Applied after the class's own sizing. + dataset_words: Option, warps: usize, nonce: u64, /// `hash-bound --count N`: N consecutive nonces from --nonce, one epoch build (gate G2, 5 October 2026). @@ -56,6 +60,12 @@ struct Args { /// generator 3 and the era bytes recorded (the measurement packs: v2's mixer under the era layout). era: Option<(u64, Vec, String)>, era_widths: Vec, + /// `--reg64`: the 64-register window per lane over the chosen class (the hash lane's measurement, 8 October + /// 2026, a research class; also `--class +reg64`). The draw is the class's; the flag is stamped on the + /// program after it, as the era is. + reg64: bool, + /// `--reg64-chain`: the full-chain address mix over the reg64 window (`+reg64c`). + reg64_chain: bool, } /// `igneum-era-test/` or `:<64 hex>` -> (index, 32 era bytes, label). @@ -80,6 +90,7 @@ fn parse_widths(s: &str) -> Option> { .map(|x| match x.trim() { "4" => Some(1u8), "16" => Some(4), + "32" => Some(8), // W = 8, the 32-byte sector (the w32 class, 8 October 2026) "64" => Some(16), _ => None, }) @@ -108,7 +119,11 @@ fn usage() -> ! { \x20 --state class v5 (or any --class ...+state): the window's state stream (IGSD1 file, igneum-day-stream --out), whose leaves key every item\n\ \x20 --shadow-reps N class v4 at a rung of the latency ladder: the shadow block's pass count (0 = the class's own 27; docs/design/latency-ladder.md), with --program-class v4\n\ \x20 --era E era layout over --class: igneum-era-test/ or :<64 hex> (the 32-byte era seed E_n)\n\ - \x20 --era-widths 4[,16,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it)" + \x20 --era-widths 4[,16,32,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it; 32 only with the w32 class)\n\ + \x20 --dataset-words N research class ds55: the dataset at N words (a multiple of 65,536 in 2^28 ..= 2^31; 1476395008 = 5.5 GiB); a non-power-of-two uses idx = (src * N) >> 32 in every load (spec 01 section 1.13.3), a power of two is --dataset-log2\n\\ + \x20 --reg64 the 64-register window per lane over the class (research; the same as --class +reg64; CUDA, OpenCL and Metal texts)\n\ + \x20 --reg64-chain reg64 with the full-chain address mix: every load's address consumes all 64 registers (the same as --class +reg64c)\n\ + \x20 --reg64-prefix export: the CUDA texts carry the reg64 chain in its closed prefix form (review B's F05; the same hash and vectors, a measurement text)" ); std::process::exit(2) } @@ -122,6 +137,7 @@ fn parse() -> Args { closed_form: false, state: None, dataset_log2: DEFAULT_DATASET_LOG2, + dataset_words: None, warps: 20, nonce: 0, count: 1, @@ -135,6 +151,8 @@ fn parse() -> Args { shadow_reps: 0, era: None, era_widths: vec![1], + reg64: false, + reg64_chain: false, }; let mut it = std::env::args().skip(1); a.cmd = it.next().unwrap_or_else(|| usage()); @@ -146,6 +164,13 @@ fn parse() -> Args { "--out" => a.out = Some(val()), "--closed-form" => a.closed_form = true, "--dataset-log2" => a.dataset_log2 = val().parse().unwrap_or_else(|_| usage()), + "--dataset-words" => { + let n: u64 = val().parse().unwrap_or_else(|_| usage()); + a.dataset_words = Some(DatasetGeom::words(n).unwrap_or_else(|err| { + eprintln!("{err}"); + std::process::exit(2) + })); + } "--warps" => a.warps = val().parse().unwrap_or_else(|_| usage()), "--nonce" => a.nonce = val().parse().unwrap_or_else(|_| usage()), "--count" => a.count = val().parse().unwrap_or_else(|_| usage()), @@ -160,6 +185,12 @@ fn parse() -> Args { "--shadow-reps" => a.shadow_reps = val().parse().unwrap_or_else(|_| usage()), "--era" => a.era = Some(parse_era(&val()).unwrap_or_else(|| usage())), "--era-widths" => a.era_widths = parse_widths(&val()).unwrap_or_else(|| usage()), + "--reg64" => a.reg64 = true, + "--reg64-chain" => { + a.reg64 = true; + a.reg64_chain = true; + } + "--reg64-prefix" => igneum_pow::emit::set_reg64_prefix(true), _ => usage(), } } @@ -200,10 +231,29 @@ fn main() { // gate G2 (5 October 2026, ported from branch ca2-era for the class v4 gate run): `--count N` prints // "nonce hash" for N consecutive 64-bit nonces from --nonce, one epoch build, to re-hash a worker's // found lines (the same prehash, target ff..ff) - for k in 0..a.count { - let n = a.nonce.wrapping_add(k); - println!("{n} {:016x}", e.hash_bound(&prehash, n)); + // one warp per 32 nonces (the P01 campaign, 8 October 2026: a million nonces per pack in minutes, not + // hours; the per-nonce form hashed the whole aligned group for every nonce) + let end = a.nonce.wrapping_add(a.count); + let mut n = a.nonce; + let mut out = String::with_capacity(1 << 16); + use std::io::Write; + let stdout = std::io::stdout(); + let mut lock = stdout.lock(); + while n != end { + let base = n & !31; + let warp = e.hash_warp_bound(&prehash, base); + let mut m = n; + while m != end && (m & !31) == base { + out.push_str(&format!("{m} {:016x}\n", warp[(m & 31) as usize])); + m = m.wrapping_add(1); + } + n = m; + if out.len() > (1 << 15) { + lock.write_all(out.as_bytes()).unwrap(); + out.clear(); + } } + lock.write_all(out.as_bytes()).unwrap(); return; } let init = igneum_pow::bind::block_init_words(&prehash, a.nonce); @@ -222,6 +272,22 @@ fn main() { fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) { let (mut e, label) = epoch_of_class(a, mode); stamp_era(&mut e, a); + // research class ds55: the dataset at the word count of --dataset-words, the cache and the items unchanged; a + // state class sizes its leaves by the power-of-two count and is refused here + if let Some(geom) = a.dataset_words { + if e.program.class.state { + eprintln!("--dataset-words is not supported with a state class ({}): the leaves are sized by --dataset-log2", e.program.class.name()); + std::process::exit(2); + } + e.dataset = e.dataset.with_geom(geom); + } + if a.reg64 { + // the 64-register window over the drawn program (the same text in CUDA, OpenCL and Metal since class-v6) + e.program.class = e.program.class.with_reg64(); + if a.reg64_chain { + e.program.class = e.program.class.with_reg64_chain(); + } + } // class v5: the leaves of --state, built for the dataset's size; a state class without --state is refused here // rather than at the first derivation if e.program.class.state { @@ -381,10 +447,10 @@ fn export(a: &Args, mode: DatasetMode) { let build_ms = t0.elapsed().as_secs_f64() * 1e3; println!("igneum-pow export {out}"); println!( - "seed \"{}\", day \"{}\", dataset 2^{} words ({}), generator v{} attempt {} program id {:016x}, loads/hash {}; epoch built in {build_ms:.1} ms", + "seed \"{}\", day \"{}\", dataset {} ({}), generator v{} attempt {} program id {:016x}, loads/hash {}; epoch built in {build_ms:.1} ms", e.program.seed_string, day_label, - e.dataset.log2_words, + e.dataset.geom.describe(), e.dataset.mode().name(), e.program.generator, e.program.attempt, @@ -392,6 +458,10 @@ fn export(a: &Args, mode: DatasetMode) { e.program.loads_per_hash() ); println!("op mix: {}; class {}, {} bytes/hash, widths (1,4,16 words) {:?}", e.program.op_mix(), e.program.class.name(), e.program.bytes_per_hash(), e.program.width_counts()); + println!("seed words {}", e.program.seed.iter().map(|w| format!("0x{w:08x}")).collect::>().join(" ")); + if e.dataset.geom.mulshift { + println!("dataset mapping: multiply-shift, idx = (src * {}) >> 32; {} items, {} bytes (research class ds55)", e.dataset.geom.words, e.dataset.geom.items(), e.dataset.geom.bytes()); + } let source = format!("igneum-pow (Rust) CPU interpreter, generator v{}, {} dataset", e.program.generator, e.dataset.mode().name()); let pack = export_pack(&e, &day_label, &source); let dir = std::path::Path::new(&out); diff --git a/igneum-pow/src/memhard.rs b/igneum-pow/src/memhard.rs index 63a35ee50..644896ffb 100644 --- a/igneum-pow/src/memhard.rs +++ b/igneum-pow/src/memhard.rs @@ -210,12 +210,15 @@ pub struct MixParams { pub redraws: u32, } -/// Class v5's mixer-draw rule (AP-F4-1): the NAF sum of the 16 multipliers at least this. -pub const MIXER_NAF_SUM_MIN: u32 = 163; -/// Class v5's mixer-draw rule: every multiplier's NAF weight at least this. +/// Class v5's mixer-draw rule (AP-F4-1, the form the attack-pass lane and adv-mixer-2 agreed on 7 October 2026, 22:0x +/// UK; it replaces the first form's "NAF sum under 163"): the adder-datapath cost of a mixer block is +/// `A = 64 + sum over the 16 multipliers of (w32(m) - 1)`, `w32` the NAF weight over bit positions 0 to 31 only (the +/// carry digit at position 32 a 32-bit multiplier never pays; the first census counted it, so its median read 231 +/// where the agreed median is 226). A block whose cost is at most this is rejected (a 1.1x gain against the median). +pub const MIXER_COST_REJECT_MAX: u32 = 205; +/// Class v5's mixer-draw rule: every multiplier's `w32` at least this (a multiplier with a two-adder chain, `w32 <= 3`, +/// is rejected on its own: adv-mixer-2's `k >= 1`). pub const MIXER_NAF_WORD_MIN: u32 = 4; -/// Class v5's mixer-draw rule: at least this many distinct rotation amounts among the eight. -pub const MIXER_DISTINCT_ROT_MIN: usize = 4; /// Class v5's mixer-draw rule: redraws before the last block stands as drawn (never reached at 6.1e-4 per try). pub const MIXER_REDRAW_CAP: u32 = 64; @@ -238,14 +241,40 @@ pub fn naf_weight(mut x: u64) -> u32 { w } -/// Whether a mixer block passes class v5's draw rule (AP-F4-1). +/// The NAF weight of a 32-bit multiplier over bit positions 0 to 31: [`naf_weight`] without the digit at position 32. +pub fn naf32_weight(m: u32) -> u32 { + let mut x = m as u64; + let mut w = 0; + let mut pos = 0; + while x != 0 { + if x & 1 == 1 { + if pos < 32 { + w += 1; + } + if x & 3 == 3 { + x += 1; + } else { + x -= 1; + } + } + x >>= 1; + pos += 1; + } + w +} + +/// The adder-datapath cost of a mixer block: `64 + sum(w32(m) - 1)` over the 16 multipliers. +pub fn mixer_cost(mul: &[u32; 16]) -> u32 { + 64 + mul.iter().map(|&m| naf32_weight(m).saturating_sub(1)).sum::() +} + +/// Whether a mixer block passes class v5's draw rule (AP-F4-1): cost over [`MIXER_COST_REJECT_MAX`], every +/// multiplier's `w32` at least [`MIXER_NAF_WORD_MIN`], and the eight rotation amounts not all equal. A rejected block +/// is redrawn whole (all forty draws) from the continuing stream. pub fn mixer_block_admissible(rot: &[u32; 8], mul: &[u32; 16]) -> bool { - let sum: u32 = mul.iter().map(|&m| naf_weight(m as u64)).sum(); - let words = mul.iter().all(|&m| naf_weight(m as u64) >= MIXER_NAF_WORD_MIN); - let mut distinct = rot.to_vec(); - distinct.sort_unstable(); - distinct.dedup(); - sum >= MIXER_NAF_SUM_MIN && words && distinct.len() >= MIXER_DISTINCT_ROT_MIN + let words = mul.iter().all(|&m| naf32_weight(m) >= MIXER_NAF_WORD_MIN); + let rot_equal = rot.iter().all(|&r| r == rot[0]); + mixer_cost(mul) > MIXER_COST_REJECT_MAX && words && !rot_equal } impl MixParams { @@ -269,11 +298,11 @@ impl MixParams { } let mut redraws = 0u32; if shape.state { - // Class v5 (docs/design/class-v5-stored-state.md section 11, AP-F4-1, the attack-pass lane's weak-day census): - // a mixer block whose multipliers are cheap on an adder datapath (NAF sum under 163, a word under NAF weight 4) - // or whose rotations repeat (under 4 distinct amounts) is redrawn from the next stream values, so no day is a - // weak day for a per-day LUT-recompute FPGA (the worst calendar day of the census, chain day 29,337, was 1.121x). - // About 6.1e-4 of days redraw. The derive program's draws (none under v5) come after, as before. + // Class v5 (docs/design/class-v5-stored-state.md section 11, AP-F4-1, the attack-pass lane's and adv-mixer-2's + // reconciled weak-day census): a mixer block whose adder cost is at most 205 against the median 226, or with a + // two-adder multiplier, or with one rotation amount, is redrawn whole from the next forty stream values, so no + // day is a weak day for a per-day LUT-recompute FPGA (the worst calendar day, chain day 29,337 = 2050-04-28, + // read 1.113x). About 5.69e-4 of days redraw, 15 days a century. The derive program's draws come after. while !mixer_block_admissible(&rot, &mul) && redraws < MIXER_REDRAW_CAP { for r in rot.iter_mut() { *r = 1 + rng.below(31) as u32; @@ -818,9 +847,11 @@ impl MemhardCpu { mod tests { use super::*; - /// Class v5's mixer-draw rule (AP-F4-1), the known-failed case first: a block of cheap multipliers (NAF sum under - /// 163) or repeated rotations is inadmissible; a scan of day keys finds days the rule redraws (the census's 6.1e-4), - /// every v5 block passes after the draw, and the v4 constants of the same keys never move. + /// Class v5's mixer-draw rule (AP-F4-1 in the agreed form), the known-failed case first: the day the two censuses + /// name as the worst of the century, chain day 29,337 (2050-04-28, 1.113x), draws a block the rule rejects and class + /// v5 redraws it; a block of two-adder multipliers or one rotation amount is inadmissible; a scan of day keys finds + /// the redrawn days (about 5.69e-4), every v5 block passes after the draw, and the v4 constants of the same keys + /// never move. #[test] fn class_v5_mixer_draw_rule() { assert_eq!(naf_weight(0), 0); @@ -828,16 +859,34 @@ mod tests { assert_eq!(naf_weight(3), 2, "11 = 100 - 1"); assert_eq!(naf_weight(7), 2, "111 = 1000 - 1"); assert_eq!(naf_weight(0xffff_ffff), 2); + assert_eq!(naf32_weight(0xffff_ffff), 1, "the carry digit at position 32 is not paid"); + assert_eq!(naf32_weight(0xc000_0001), 2, "2^32 - 2^30 + 1: the digit at 32 dropped, -2^30 and +1 kept"); assert_eq!(naf_weight(0b1010_1010), 4); + assert_eq!(naf32_weight(0b1010_1010), 4); let good_rot = [1u32, 5, 9, 13, 17, 21, 25, 29]; let cheap = [0x8000_0001u32; 16]; + assert_eq!(mixer_cost(&cheap), 64 + 16, "16 two-adder multipliers"); assert!(!mixer_block_admissible(&good_rot, &cheap), "the known-failed case: 16 two-adder multipliers"); let dense = [0xaaaa_aaabu32; 16]; + assert!(mixer_cost(&dense) > MIXER_COST_REJECT_MAX); assert!(mixer_block_admissible(&good_rot, &dense)); assert!(!mixer_block_admissible(&[7u32; 8], &dense), "one rotation amount"); - assert!(!mixer_block_admissible(&[1u32, 2, 3, 3, 3, 3, 3, 3], &dense), "three distinct amounts"); + assert!(mixer_block_admissible(&[1u32, 2, 3, 3, 3, 3, 3, 3], &dense), "three distinct amounts pass the agreed form"); + // one two-adder multiplier among dense ones is rejected on its own (k >= 1) + let mut one_cheap = dense; + one_cheap[5] = 0x0000_0401; + assert!(mixer_cost(&one_cheap) > MIXER_COST_REJECT_MAX); + assert!(!mixer_block_admissible(&good_rot, &one_cheap), "a two-adder multiplier"); + // the known-failed day let v5 = Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: true }; let v4 = Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: false }; + let worst = crate::seed::seed_words_from_bytes(&crate::bind::day_bytes(29_337)); + let b = MixParams::with_shape(worst, v4); + println!("day 29,337 (2050-04-28) under class v4: cost {}, w32 min {}, distinct rot {}", mixer_cost(&b.mul), b.mul.iter().map(|&m| naf32_weight(m)).min().unwrap(), { let mut r = b.rot.to_vec(); r.sort_unstable(); r.dedup(); r.len() }); + assert!(!mixer_block_admissible(&b.rot, &b.mul), "the known-failed day: the sub-version 3 block of day 29,337 is one the rule rejects"); + let a = MixParams::with_shape(worst, v5); + assert!(a.redraws >= 1 && mixer_block_admissible(&a.rot, &a.mul), "class v5 redraws day 29,337"); + assert_ne!((a.rot, a.mul), (b.rot, b.mul)); let mut redrawn = 0; let mut scanned = 0; for d in 0..60_000u64 { @@ -859,7 +908,7 @@ mod tests { break; } } - assert!(redrawn >= 1, "no redraw in {scanned} days (the census says about 6.1e-4 per day)"); + assert!(redrawn >= 1, "no redraw in {scanned} days (the census says about 5.69e-4 per day)"); } #[test] diff --git a/igneum-pow/src/packcheck.rs b/igneum-pow/src/packcheck.rs index a7cbea7f5..b75e5a073 100644 --- a/igneum-pow/src/packcheck.rs +++ b/igneum-pow/src/packcheck.rs @@ -199,14 +199,14 @@ pub fn verify_pack_texts_chain( // carries IGNEUM_STATE_ROOT_HEX (and the shadow block of v4) and no other generator does. let state_root_hex = define_str(program_h, "IGNEUM_STATE_ROOT_HEX"); match (class, state_root_hex.is_some()) { - (ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_STATE_ROOT_HEX: not a class v5 pack".into())), - (ProgramClass::V5, true) => {} + (ProgramClass::V5 | ProgramClass::V6, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 or 6 (a state class) without IGNEUM_STATE_ROOT_HEX: not a class v5 pack".into())), + (ProgramClass::V5 | ProgramClass::V6, true) => {} (_, true) => return Err(PackFault::Disagree(format!("IGNEUM_GENERATOR {generator} with class v5 state lines: a program over state leaves is generator 5 (export the pack as class v5)"))), _ => {} } match (class, shadow > 0) { (ProgramClass::V4, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 4 (class v4) without IGNEUM_SHADOW_INSTRS: not a class v4 pack".into())), - (ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_SHADOW_INSTRS: not a class v5 pack".into())), + (ProgramClass::V5 | ProgramClass::V6, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 or 6 (a state class) without IGNEUM_SHADOW_INSTRS: not a class v5 pack".into())), // the class v4 stream sub-version (AP-F8-1 amendment, 7 October 2026): a generator 4 pack from before the // load-source rule carries no IGNEUM_PROGRAM_SUBVERSION and its program id is another stream's; refused (ProgramClass::V4, true) if define_u32(program_h, "IGNEUM_PROGRAM_SUBVERSION") != Some(u32::from(crate::generator::PROGRAM_SUBVERSION_V4)) => { diff --git a/igneum-pow/src/verify.rs b/igneum-pow/src/verify.rs index 9d1281aeb..c0dc217b8 100644 --- a/igneum-pow/src/verify.rs +++ b/igneum-pow/src/verify.rs @@ -9,18 +9,38 @@ use crate::seed::day_key; /// window of the load site, `k = min(win, D - 26)` (0 when `D <= 26`), `idx = ((y & (MASK >> k)) | ((off & /// (2^k - 1)) << (D - k))) & MASK`. For every other class `idx = x & MASK`, the lottery hash's address. `mask` is /// `2^D - 1`. The acceptance mirror calls this at the rule's constant `D = 28`. +/// +/// Class v6 lane 1, the index fold (`docs/design/class-v6-rotating-family.md` section 2; `EraParams::fold`): the +/// product's low bits are folded before the rotation, `y = x * M; y ^= y >> 16; y = rotl(y, R)` ([`stride`]), so no +/// era's R lands a biased product bit (bit 0 of `x * M` for odd M is bit 0 of x; bit 1 is set at 3/8) on an address +/// bit. Every era of every other class keeps the plain product. #[inline(always)] pub fn load_index(era: Option<&EraParams>, ins: &Instr, x: u32, mask: u32, log2: u32) -> u32 { match era { None => x & mask, Some(e) => { let (wm, off) = window(ins, mask, log2); - let y = x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot); + let y = stride(e, x); ((y & wm) | off) & mask } } } +/// The era's stride of a source value: `rotl(x * M, R)`, with the index fold of class v6 lane 1 between the product +/// and the rotation when the era carries it (`y ^= y >> 16`). The one place the form lives on the CPU side; the three +/// emitters write the same text ([`crate::emit`]). +#[inline(always)] +pub fn stride(e: &EraParams, x: u32) -> u32 { + let mut y = x.wrapping_mul(e.stride_mul); + if e.fold { + y ^= y >> 16; + } + y.rotate_left(e.stride_rot) +} + +/// The fold's shift (`y ^= y >> INDEX_FOLD_SHIFT`), named for the pack texts and the tests. +pub const INDEX_FOLD_SHIFT: u32 = 16; + /// The window of a load site at a dataset of `2^log2` words: `(window mask, offset)` such that /// `idx = (y & window mask) | offset` lies in the site's aligned window of `2^(log2 - k)` words. #[inline(always)] @@ -31,6 +51,122 @@ pub fn window(ins: &Instr, mask: u32, log2: u32) -> (u32, u32) { (wm, off) } +/// The window of a load site in the 32-bit SOURCE space (the multiply-shift mapping, research class ds55, +/// 8 October 2026): `k = min(win, log2 - 26)` as [`window`] with `log2 = floor(log2(N))`, and `(window mask, +/// offset)` such that `v = (y & window mask) | offset` lies in the site's aligned window of `2^(32 - k)` source +/// values; `idx = (v * N) >> 32` then lands in a contiguous run of about `N / 2^k` words, the site's window of the +/// dataset. +#[inline(always)] +pub fn window32(ins: &Instr, log2: u32) -> (u32, u32) { + let k = (ins.win as u32).min(log2.saturating_sub(26)); + let wm = u32::MAX >> k; + let off = ((((ins.off as u32) & ((1u32 << k) - 1)) as u64) << (32 - k)) as u32; + (wm, off) +} + +/// The dataset's geometry: `2^log2` words under the lottery hash's `src AND MASK` (`mulshift` false), or `words` +/// words, any multiple of 2^16 in `2^28 ..= 2^31`, under the multiply-shift range reduction of spec 01 section +/// 1.13.3 (`mulshift` true: `idx = (src * words) >> 32` in 64 bits; research class ds55, 8 October 2026, no +/// consensus object moves). A power of two given as a word count takes the mask path, so `--dataset-words 2^28` +/// is `--dataset-log2 28` byte for byte. `log2` is `floor(log2(words))` under the multiply-shift (the era +/// window's floor rule reads it); the item index `words / 16 - 1` stays 32-bit. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct DatasetGeom { + pub log2: u32, + pub words: u64, + pub mulshift: bool, +} + +impl DatasetGeom { + /// The smallest word count `--dataset-words` takes: 2^28 (1 GiB). + pub const MIN_WORDS: u64 = 1 << 28; + /// The largest: 2^31 (8 GiB; the item index stays 32-bit far beyond, the kernel's word index is 32-bit). + pub const MAX_WORDS: u64 = 1 << 31; + /// The step: 2^16 words, so the era layout's interleave (positions below 16) stays a bijection of the dataset + /// and every wide or warp-coalesced load stays inside it. + pub const WORDS_STEP: u64 = 1 << 16; + + /// A dataset of `2^log2` words (4 to 32), the lottery hash's mask path. + pub fn pow2(log2: u32) -> Self { + assert!((4..=32).contains(&log2), "dataset log2 must be in 4..=32"); + Self { log2, words: 1u64 << log2, mulshift: false } + } + + /// A dataset of `words` words: the mask path for a power of two, the multiply-shift otherwise. Refused + /// outside `MIN_WORDS ..= MAX_WORDS` or off the `WORDS_STEP` grid, with the reason. + pub fn words(words: u64) -> Result { + if !(Self::MIN_WORDS..=Self::MAX_WORDS).contains(&words) { + return Err(format!( + "--dataset-words {words}: the word count must be in 2^28 ..= 2^31 ({} ..= {})", + Self::MIN_WORDS, + Self::MAX_WORDS + )); + } + if words % Self::WORDS_STEP != 0 { + return Err(format!("--dataset-words {words}: the word count must be a multiple of 2^16 words (65,536; the era interleave and the wide loads)")); + } + if words.is_power_of_two() { + return Ok(Self::pow2(words.trailing_zeros())); + } + Ok(Self { log2: words.ilog2(), words, mulshift: true }) + } + + /// The last word index (`words - 1`). Under the mask path it is the AND mask. + pub fn last_index(&self) -> u32 { + (self.words - 1) as u32 + } + + /// The AND mask of the mask path; under the multiply-shift the last index (recorded in packs as `IGNEUM_MASK` + /// for the host's self-test, never ANDed). + pub fn mask(&self) -> u32 { + self.last_index() + } + + pub fn items(&self) -> u64 { + self.words >> 4 + } + + pub fn bytes(&self) -> u64 { + self.words << 2 + } + + /// The range reduction of a 32-bit source value to a word index. + #[inline(always)] + pub fn reduce(&self, x: u32) -> u32 { + if self.mulshift { + ((x as u64 * self.words) >> 32) as u32 + } else { + x & self.last_index() + } + } + + /// One line for logs: `2^28 words` or `1476395008 words (5.50 GiB, not a power of two, multiply-shift)`. + pub fn describe(&self) -> String { + if self.mulshift { + format!("{} words ({:.2} GiB, not a power of two, multiply-shift)", self.words, self.bytes() as f64 / (1u64 << 30) as f64) + } else { + format!("2^{} words", self.log2) + } + } +} + +/// [`load_index`] at a dataset geometry: the mask path unchanged; under the multiply-shift the plain load is +/// `(x * N) >> 32` and the era form windows the source first ([`window32`]) then reduces. +#[inline(always)] +pub fn load_index_geom(era: Option<&EraParams>, ins: &Instr, x: u32, geom: DatasetGeom) -> u32 { + if !geom.mulshift { + return load_index(era, ins, x, geom.mask(), geom.log2); + } + match era { + None => geom.reduce(x), + Some(e) => { + let (wm, off) = window32(ins, geom.log2); + let y = stride(e, x); + geom.reduce((y & wm) | off) + } + } +} + /// Read-width experiment (5 October 2026): a `load` of `W` words folds every word into `dst`: /// `x = dst XOR w[0]; for j in 1..W: x = (rotl(x, FOLD_ROT) * FOLD_MUL) XOR w[j]; dst = x`. For `W = 1` this is the /// lottery hash's `dst XOR dataset[...]`. The fold is state-dependent (the rotate-multiply sits between the words), @@ -184,8 +320,12 @@ pub enum Dataset { /// A dataset of `2^log2` words plus the construction that fills it. pub struct DatasetSource { + /// `floor(log2(words))`: the size under the mask path, the era window's floor under the multiply-shift. pub log2_words: u32, + /// The last word index: the AND mask under the mask path (see [`DatasetGeom::mask`]). pub mask: u32, + /// The geometry (size and range reduction); `log2_words` and `mask` are its `log2` and `mask()`. + pub geom: DatasetGeom, /// The day key `K`; `d0, d1 = K[0], K[1]`. pub key: [u32; 8], /// The bytes `K` was derived from (`"day/"` for a string day, `bind::day_bytes` on the chain), recorded @@ -219,12 +359,25 @@ impl DatasetSource { /// `2^shape.cache_log2_words` words on the calling thread. pub fn from_key_shape(key: [u32; 8], mode: DatasetMode, log2_words: u32, shape: Shape) -> Self { assert!((4..=32).contains(&log2_words), "dataset log2 must be in 4..=32"); - let mask = if log2_words == 32 { u32::MAX } else { (1u32 << log2_words) - 1 }; + Self::from_key_geom(key, mode, DatasetGeom::pow2(log2_words), shape) + } + + /// [`DatasetSource::from_key_shape`] at a dataset geometry (the multiply-shift sizes of `--dataset-words`). + pub fn from_key_geom(key: [u32; 8], mode: DatasetMode, geom: DatasetGeom, shape: Shape) -> Self { let dataset = match mode { DatasetMode::ClosedForm => Dataset::ClosedForm { d0: key[0], d1: key[1] }, DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::with_shape(key, shape)), }; - Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None } + Self { log2_words: geom.log2, mask: geom.mask(), geom, key, key_bytes: Vec::new(), dataset, hot: None } + } + + /// This source at another geometry: the cache, key and construction unchanged (an item has the same value at + /// every size), only the size and the range reduction move. + pub fn with_geom(mut self, geom: DatasetGeom) -> Self { + self.log2_words = geom.log2; + self.mask = geom.mask(); + self.geom = geom; + self } /// This source with the window's state leaves (class v5, `docs/design/class-v5-stored-state.md`): memory-hard mode @@ -246,7 +399,7 @@ impl DatasetSource { Dataset::MemoryHard(m) => Dataset::MemoryHard(m.refreshed(leaves)), Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"), }; - Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } + Self { log2_words: self.log2_words, mask: self.mask, geom: self.geom, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } } /// A copy of this source sharing its cache (and leaves), for a caller that needs an owned source from a shared one. @@ -255,7 +408,7 @@ impl DatasetSource { Dataset::MemoryHard(m) => Dataset::MemoryHard(crate::memhard::MemhardCpu { params: m.params.clone(), cache: m.cache.clone(), leaves: m.leaves.clone() }), Dataset::ClosedForm { d0, d1 } => Dataset::ClosedForm { d0: *d0, d1: *d1 }, }; - Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } + Self { log2_words: self.log2_words, mask: self.mask, geom: self.geom, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } } /// The window's state leaves, when the source carries them. @@ -299,8 +452,14 @@ impl DatasetSource { } /// `dataset[w & mask]` under a program's layout (era layout). The closed form has no items and ignores it. + /// Under the multiply-shift geometry `w` must already be a word index (below `words`). pub fn word_at(&self, layout: Layout, w: u32) -> u32 { - let w = w & self.mask; + let w = if self.geom.mulshift { + assert!((w as u64) < self.geom.words, "word {w} outside a dataset of {} words", self.geom.words); + w + } else { + w & self.mask + }; match &self.dataset { Dataset::ClosedForm { d0, d1 } => dataset_elem(w, *d0, *d1), Dataset::MemoryHard(m) => m.word_at(layout, w), @@ -375,11 +534,38 @@ pub fn interpret_warp_scratch( ds: &DatasetSource, trace: bool, ) -> (WarpResult, Vec) { - let mask = ds.mask; - let log2 = ds.log2_words; + interpret_warp_core(program, seed, base_nonce, ds, trace, None) +} + +/// The liveness probe of the reg64 window (`accept::check_window_liveness`): `flip` xors register `k` of every +/// lane with `x` at the start of iteration 0, after the init; `first_load_idx` receives the word index each +/// lane read at iteration 0's first load. +#[derive(Clone, Debug, Default)] +pub struct Probe { + pub flip: Option<(usize, u32)>, + pub first_load_idx: [u32; LANES], + pub seen_first_load: bool, +} + +/// [`interpret_warp_init`] under a [`Probe`] (the reg64 liveness rule): the same interpreter, one flip, one read. +pub fn interpret_warp_probe(program: &Program, seed: &[u32; 8], base_nonce: u32, ds: &DatasetSource, probe: &mut Probe) -> WarpResult { + interpret_warp_core(program, seed, base_nonce, ds, false, Some(probe)).0 +} + +fn interpret_warp_core( + program: &Program, + seed: &[u32; 8], + base_nonce: u32, + ds: &DatasetSource, + trace: bool, + mut probe: Option<&mut Probe>, +) -> (WarpResult, Vec) { + let geom = ds.geom; let era = program.class.era; let layout = program.class.layout(); - let mut r = [[0u32; LANES]; 8]; + // 8 registers per lane, or 64 under the reg64 flag (r8..r63 derived from r0..r7 as the kernels do it) + let nregs = program.registers(); + let mut r = vec![[0u32; LANES]; nregs]; for lane in 0..LANES { let nonce = base_nonce.wrapping_add(lane as u32); for i in 0..8 { @@ -388,7 +574,13 @@ pub fn interpret_warp_scratch( x = splitmix32(x); r[i][lane] = x ^ seed[(i + 1) & 7]; } + for k in 8..nregs { + r[k][lane] = r[k & 7][lane].wrapping_mul(0x9e3779b9u32).wrapping_add(k as u32); + } } + // the iteration's statements: the drawn program, or its interleaved two-window form under reg64 + let scheduled = program.scheduled(); + let address_mix = program.address_mix(); let mut items_derived = 0usize; let mut idx = [0u32; LANES]; let mut val = [0u32; LANES]; @@ -403,10 +595,28 @@ pub fn interpret_warp_scratch( let h = ds.hot.as_ref().expect("a hot-table program needs the epoch's hot table on the dataset source"); assert_eq!(h.n_words(), program.hot_words(), "the hot table's size is the class's"); } - for _ in 0..ITERATIONS { + for it in 0..ITERATIONS { + if it == 0 { + // the liveness probe's flip: one register of every lane, complemented at the start of iteration 0 + if let Some(pr) = probe.as_deref_mut() { + if let Some((k, x)) = pr.flip { + for lane in 0..LANES { + r[k][lane] ^= x; + } + } + } + } let sel = r[0]; - for ins in &program.instrs { - step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived); + for ins in &scheduled { + step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix); + if it == 0 && ins.op == Op::Load { + if let Some(pr) = probe.as_deref_mut() { + if !pr.seen_first_load { + pr.first_load_idx = idx; + pr.seen_first_load = true; + } + } + } if ins.op == Op::Scratch { let m = scratch.as_mut().expect("a scratch op needs a scratch class"); let (d, a) = (ins.dst as usize, ins.src as usize); @@ -420,7 +630,17 @@ pub fn interpret_warp_scratch( // iteration's `sel`; it is empty on every class without a shadow, so version 2 and class v3 run nothing here. for _ in 0..program.shadow_reps() { for ins in &program.shadow { - step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived); + step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix); + } + } + } + if nregs > 8 { + // reg64: both windows fold into the eight output registers by xor before the hash fold + for k in 0..8 { + for j in (k + 8..nregs).step_by(8) { + for lane in 0..LANES { + r[k][lane] ^= r[j][lane]; + } } } } @@ -438,19 +658,38 @@ pub fn interpret_warp_scratch( #[allow(clippy::too_many_arguments)] fn step( ins: &Instr, - r: &mut [[u32; LANES]; 8], + r: &mut [[u32; LANES]], sel: &[u32; LANES], - mask: u32, - log2: u32, + geom: DatasetGeom, era: Option<&EraParams>, layout: Layout, ds: &DatasetSource, idx: &mut [u32; LANES], val: &mut [u32; LANES], items_derived: &mut usize, + address_mix: bool, ) { let d = ins.dst as usize; let a = ins.src as usize; + // reg64 full chain: a load's address source is src ^ m, m the rotate-xor chain over the 63 other registers in + // index order (m = first; m = rotl(m, 1) ^ next). The source itself stays out of the chain: with it inside, a + // register whose term lands at rotation 0 mod 32 (r31, r63) cancels its own direct term, a dead register the + // liveness rule found on the pinned draw (first load src r31, 8 October 2026, 15:5x UK). + let addr_src = |r: &[[u32; LANES]], lane: usize| -> u32 { + if !address_mix { + return r[a][lane]; + } + let mut m = 0u32; + let mut started = false; + for k in 0..r.len() { + if k == a { + continue; + } + m = if started { m.rotate_left(1) ^ r[k][lane] } else { r[k][lane] }; + started = true; + } + r[a][lane] ^ m + }; match ins.op { Op::Add => { let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32); @@ -519,7 +758,7 @@ fn step( } Op::Load if ins.width == 1 => { for lane in 0..LANES { - idx[lane] = load_index(era, ins, r[a][lane], mask, log2); + idx[lane] = load_index_geom(era, ins, addr_src(r, lane), geom); } *items_derived += ds.fetch(idx, val, layout); for lane in 0..LANES { @@ -531,7 +770,7 @@ fn step( let width = ins.width as usize; let align = !(ins.width as u32 - 1); for lane in 0..LANES { - idx[lane] = load_index(era, ins, r[a][lane], mask, log2) & align; + idx[lane] = load_index_geom(era, ins, addr_src(r, lane), geom) & align; } let mut vals = [[0u32; 16]; LANES]; *items_derived += ds.fetch_wide(idx, width, &mut vals, layout); @@ -551,8 +790,8 @@ fn step( } } Op::WLoad => { - // Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l. - let base = (r[a][0] & mask) & !31; + // Lane 0's register, range-reduced, aligned down to 32 words; lane l reads word base + l. + let base = geom.reduce(r[a][0]) & !31; for lane in 0..LANES { idx[lane] = base + lane as u32; } @@ -933,3 +1172,91 @@ mod tests { assert!(!e.verify_block(0, w[0] - 1)); } } + + +/// Review B's F05 (8 October 2026): the reg64 full-chain address source in closed form. The reference fold +/// (`m = r[k0]; m = rotl(m, 1) ^ r[k1]; ...` over the 63 registers other than `s`, in index order) equals +/// `r[s] ^ ror(P_s, 1) ^ (S ^ P_s ^ a[s])` with `a[k] = rotl(r[k], (63 - k) mod 32)`, `S` the xor of every `a[k]` +/// and `P_s` the xor of `a[k]` for `k < s`: a prefix-xor structure answers every load from one `S` and one prefix, +/// which is the incremental form a chip would keep. The verifier keeps the fold as the definition; this form is the +/// test's and the fleet's measurement's. +pub fn reg64_address_source(r: &[u32; 64], s: usize) -> u32 { + let a = |k: usize| r[k].rotate_left(((63 - k) % 32) as u32); + let mut total = 0u32; + let mut prefix = 0u32; + for k in 0..64 { + total ^= a(k); + if k < s { + prefix ^= a(k); + } + } + r[s] ^ prefix.rotate_right(1) ^ (total ^ prefix ^ a(s)) +} + +/// The reference fold of [`reg64_address_source`] on one lane's registers, as `interpret_warp_core` computes it. +pub fn reg64_address_source_reference(r: &[u32; 64], s: usize) -> u32 { + let mut m = 0u32; + let mut started = false; + for (k, &v) in r.iter().enumerate() { + if k == s { + continue; + } + m = if started { m.rotate_left(1) ^ v } else { v }; + started = true; + } + r[s] ^ m +} + +#[cfg(test)] +mod reg64_mix_tests { + use super::*; + + /// Review B's F05, known-failed first: a form with one rotation off (every term rotated as if it sat after the + /// source) disagrees with the reference on some source; the closed form agrees on every source register over + /// 256 register states drawn from a fixed stream and over the states a real program leaves in its registers. + #[test] + fn reg64_closed_form_equals_the_reference_fold_on_every_source() { + let wrong = |r: &[u32; 64], s: usize| -> u32 { + let a = |k: usize| r[k].rotate_left(((63 - k) % 32) as u32); + let total: u32 = (0..64).map(a).fold(0, |x, y| x ^ y); + r[s] ^ total ^ a(s) + }; + let mut x = 0x9e37_79b9_7f4a_7c15u64; + let mut next = || { + x = x.wrapping_add(0x9e37_79b9_7f4a_7c15); + let mut z = x; + z = (z ^ (z >> 30)).wrapping_mul(0xbf58_476d_1ce4_e5b9); + z = (z ^ (z >> 27)).wrapping_mul(0x94d0_49bb_1331_11eb); + (z ^ (z >> 31)) as u32 + }; + let mut wrong_differs = false; + for _ in 0..256 { + let mut r = [0u32; 64]; + for v in r.iter_mut() { + *v = next(); + } + for s in 0..64 { + assert_eq!(reg64_address_source(&r, s), reg64_address_source_reference(&r, s), "source r{s}"); + if wrong(&r, s) != reg64_address_source_reference(&r, s) { + wrong_differs = true; + } + } + } + assert!(wrong_differs, "the known-failed form must disagree somewhere"); + // the init rule's registers (r0..r7 seeded, r[k] = r[k & 7] * 0x9e3779b9 + k) and a few real updates + let mut r = [0u32; 64]; + for (k, v) in r.iter_mut().enumerate().take(8) { + *v = next() ^ (k as u32); + } + for k in 8..64 { + r[k] = r[k & 7].wrapping_mul(0x9e37_79b9).wrapping_add(k as u32); + } + for step in 0..64 { + let d = (step * 7 + 3) % 64; + r[d] = r[d].wrapping_mul(r[(d + 13) % 64]).wrapping_add(next()); + for s in 0..64 { + assert_eq!(reg64_address_source(&r, s), reg64_address_source_reference(&r, s), "step {step} source r{s}"); + } + } + } +} diff --git a/igneum-pow/tests/mixer.rs b/igneum-pow/tests/mixer.rs index f4644ae8d..43ddeeb41 100644 --- a/igneum-pow/tests/mixer.rs +++ b/igneum-pow/tests/mixer.rs @@ -347,8 +347,8 @@ fn determinism_v3() { }; println!("determinism {}: two builds equal, against the pinned pack {}", class.name(), dir.display()); for (name, text) in &pa.files { - if name == "vectors.json" || name == "vectors.h" { - continue; // the source string differs ("a" here) + if name == "vectors.json" || name == "vectors.h" || name == "identity.json" { + continue; // the source string differs ("a" here); identity.json is new beside the pinned twelve (A06) } let on_disk = std::fs::read_to_string(dir.join(name)).unwrap(); assert_eq!(&on_disk, text, "{name}"); diff --git a/igneum-pow/tests/packs.rs b/igneum-pow/tests/packs.rs index f66d91536..cd872c6a5 100644 --- a/igneum-pow/tests/packs.rs +++ b/igneum-pow/tests/packs.rs @@ -433,7 +433,9 @@ fn check_export(pack: &str) { if e.dataset.mode() == DatasetMode::MemoryHard { expected.extend(["memhard.h", "memhard.metal"]); } - assert_eq!(out.files.iter().map(|(n, _)| n.as_str()).collect::>(), expected); + let mut with_identity: Vec<&str> = expected.clone(); + with_identity.insert(1, "identity.json"); + assert_eq!(out.files.iter().map(|(n, _)| n.as_str()).collect::>(), with_identity); let mut on_disk: Vec = std::fs::read_dir(pack_dir(pack)) .unwrap() .map(|d| d.unwrap().file_name().to_string_lossy().to_string()) @@ -646,6 +648,9 @@ fn era_emitted_sources_match_and_loads_have_the_era_form() { let source = era_json(pack, "vectors.json")["source"].as_str().unwrap().to_string(); let out = export_pack(e, &day, &source); for (name, text) in &out.files { + if name == "identity.json" { + continue; // A06's identity file is new beside the pinned files + } let want = era_read(pack, name); assert!(text == &want, "{pack}/{name} differs from the emitter"); } @@ -655,7 +660,7 @@ fn era_emitted_sources_match_and_loads_have_the_era_form() { .filter(|n| !n.starts_with('.') && n != "seeds.txt") .collect(); on_disk.sort(); - let mut want: Vec = out.files.iter().map(|(n, _)| n.clone()).collect(); + let mut want: Vec = out.files.iter().map(|(n, _)| n.clone()).filter(|n| n != "identity.json").collect(); want.sort(); assert_eq!(on_disk, want, "{pack}: the pack holds the export's files and seeds.txt only"); let era = e.program.class.era.unwrap(); @@ -869,7 +874,7 @@ fn hot_packs_emitted_sources_and_load_forms() { let file = |name: &str| -> &str { &out.files.iter().find(|(n, _)| n == name).unwrap().1 }; hassert_same_text(pack, "vectors.json", file("vectors.json")); hassert_same_text(pack, "vectors.h", file("vectors.h")); - assert_eq!(out.files.len(), 12); + assert_eq!(out.files.len(), 13); // One form per dialect, exactly 16 - k masked dataset loads and k hot loads in every hash kernel; the fill // kernel is present once per source that builds the table. for (file, load, masked, hot) in [ @@ -974,6 +979,9 @@ fn v5_pack_is_the_v4_program_over_the_state_leaves() { let v = v5_json(pack, "vectors.json"); let out = export_pack(e, v["day"].as_str().unwrap(), v["source"].as_str().unwrap()); for (name, text) in &out.files { + if name == "identity.json" { + continue; // A06's identity file is new beside the pinned files + } assert_eq!(v5_read(pack, name), *text, "{pack}/{name} differs from the export"); } for (name, bytes) in &out.binaries { @@ -997,3 +1005,292 @@ fn v5_pack_is_the_v4_program_over_the_state_leaves() { let mh5 = v5_read("v5-genesis", "memhard.h"); assert!(mh5.contains("s[i] ^= leaf[i]"), "the leaf XOR before the first mixer"); } + +// --------------------------------------------------------------------------------------------------------------- +// The 64-register window (the hash lane's measurement, 8 October 2026, a research class behind `+reg64`). + +/// The `source` line the pinned devnet pack was exported with (vectors.json and vectors.h carry it). +const PINNED_SOURCE: &str = "igneum-pow (Rust) CPU interpreter, generator v3, memory-hard dataset"; + +/// The pinned devnet pack's epoch again, with the reg64 flag stamped on the program after the draw (as `--reg64` +/// does): the same instructions, the same dataset, 64 registers per lane. +fn epoch_reg64_of_devnet() -> Epoch { + let pack = "mx8-devnet-epoch0"; + let j = json(pack, "program.json"); + let seed = j["seed"].as_str().unwrap(); + let seed_bytes = unhex(&j["seed_bytes"]); + let day_bytes = unhex(&j["dataset"]["day_bytes"]); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + let era = j.get("era_seed_bytes").map(unhex); + let mut program = generate_from_seed_bytes_program_class(seed, &seed_bytes, ProgramClass::V3, era.as_deref()); + program.class = program.class.with_reg64(); + let shape = Shape::for_class(&program.class); + let mut dataset = DatasetSource::from_key_shape(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2, shape); + dataset.key_bytes = day_bytes; + Epoch { program, dataset } +} + +/// The plain path re-exports the pinned devnet pack unchanged: every file `export` writes, byte for byte (the +/// reg64 code sits behind the flag; a pack without it does not move). +#[test] +fn reg64_plain_path_reexports_the_pinned_devnet_pack_unchanged() { + let pack = "mx8-devnet-epoch0"; + let e = epoch(pack); + assert!(!e.program.class.reg64); + assert_eq!(e.program.registers(), 8); + assert_eq!(e.program.scheduled(), e.program.instrs, "without the flag the schedule is the drawn program"); + let out = export_pack(e, &day_label(pack), PINNED_SOURCE); + assert_eq!(out.files.len(), 13); + for (name, text) in &out.files { + if name == "identity.json" { + continue; // A06's identity file is new beside the pinned twelve + } + assert_same_text(pack, name, text); + } +} + +/// The reg64 variant: the CPU verifier and the emitted CUDA text read one schedule, so the kernel's statements name +/// the registers the verifier wrote, in the verifier's order; the vectors are the verifier's; the plain pack's +/// instructions are untouched. +#[test] +fn reg64_cpu_verifier_and_cuda_text_agree() { + let plain = epoch("mx8-devnet-epoch0"); + let e = epoch_reg64_of_devnet(); + let p = &e.program; + assert!(p.class.reg64); + assert_eq!(p.registers(), 64); + assert_eq!(p.class.name(), "mx8-erad810f22d+reg64", "the pinned pack's era class, the window outermost"); + assert_eq!(LoadClass::parse("mx8+reg64"), Some(V3_CLASS.with_reg64()), "the class name parses"); + assert_eq!(LoadClass::MX8.with_reg64().name(), "mx8+reg64", "and round-trips"); + assert_eq!(p.instrs, plain.program.instrs, "the draw is the class's without the flag"); + assert_ne!(p.program_id(), plain.program.program_id(), "the id carries the flag"); + // the schedule: 128 statements, instruction i of the draw on window 0 (+8 * (i % 4)) then on window 1 (+32) + let sched = p.scheduled(); + assert_eq!(sched.len(), 128); + let mut touched = [false; 64]; + for (i, ins) in p.instrs.iter().enumerate() { + let off = (8 * (i % 4)) as u8; + let a = sched[2 * i]; + let b = sched[2 * i + 1]; + assert_eq!((a.op, a.dst, a.src, a.src2), (ins.op, ins.dst + off, ins.src + off, ins.src2 + off), "instruction {i} on window 0"); + assert_eq!((b.op, b.dst, b.src, b.src2), (ins.op, a.dst + 32, a.src + 32, a.src2 + 32), "instruction {i} on window 1"); + touched[a.dst as usize] = true; + touched[b.dst as usize] = true; + } + // the fixed extension 8 * (i % 4) gives each octet 16 of the 64 instructions, so a draw need not write every + // register: the pinned draw writes 28 of 32 per window (r11, r17, r24, r29 unwritten, read or folded only) + let written = touched.iter().filter(|&&t| t).count(); + assert!(written >= 48 && written % 2 == 0, "the schedule writes {written} of 64 registers (both windows alike)"); + // the CUDA texts: 64 registers declared, 56 derived, every statement of the schedule in order, the fold, 32 masked loads + let mp = e.dataset.memhard().map(|m| &m.params); + for (name, text) in [("kernel.cu", cuda_kernel(p, mp)), ("kernel_bound.cu", cuda_kernel_bound(p, mp))] { + assert!(text.contains("uint32_t r0, r1, r2, r3, r4, r5, r6, r7, r8, r9,") && text.contains(", r62, r63;"), "{name}: 64 registers"); + for k in 8..64 { + assert!(text.contains(&format!(" r{k} = r{} * 0x9e3779b9u + {k}u;\n", k & 7)), "{name}: r{k} derived"); + } + for k in 0..8 { + let terms: Vec = (k + 8..64).step_by(8).map(|j| format!("r{j}")).collect(); + assert!(text.contains(&format!(" r{k} = r{k} ^ {};\n", terms.join(" ^ "))), "{name}: the fold of r{k}"); + } + assert_eq!(text.matches("ds[((rotl_imm(").count(), 2 * LOAD_SLOTS, "{name}: 32 loads (the era form)"); + assert_eq!(text.matches(" & mask]").count(), 2 * LOAD_SLOTS, "{name}: 32 masked loads"); + // every statement line names the schedule's registers: dst first, then src on every op that reads one + let lines: Vec<&str> = text.lines().filter(|l| l.starts_with(" r") && l.contains(" // ")).collect(); + assert_eq!(lines.len(), 128, "{name}: 128 statements per iteration"); + for (k, (line, ins)) in lines.iter().zip(sched.iter()).enumerate() { + let stmt = line.trim_start(); + assert!(stmt.starts_with(&format!("r{} = r{} ", ins.dst, ins.dst)) || stmt.starts_with(&format!("r{} = ", ins.dst)), "{name} statement {k}: dst r{}: {stmt}", ins.dst); + assert!(line.ends_with(&format!("// {k} {}", ins.op.name())), "{name} statement {k}: numbered: {line}"); + if ins.op != Op::Rotl && ins.op != Op::Load { + assert!(stmt.contains(&format!("r{}", ins.src)) || stmt.contains(&format!("r{},", ins.src)), "{name} statement {k}: src r{}: {stmt}", ins.src); + } + } + } + // Metal and OpenCL carry the same window (class-v6, 8 October 2026): 64 registers declared, 56 derived, the fold, + // 128 statements per iteration in the schedule's order, 32 loads of the era form + for (name, text) in [ + ("program.metal", metal_program(p, e.dataset.log2_words, LoadSource::Stored)), + ("program_bound.metal", metal_program_bound(p, e.dataset.log2_words)), + ("kernel.cl", opencl_kernel(p, mp)), + ("kernel_bound.cl", opencl_kernel_bound(p, mp)), + ] { + assert!(!text.starts_with("// reg64: no "), "{name}: the window text, not a stub"); + assert!(text.contains("uint r0, r1, r2, r3, r4, r5, r6, r7, r8, r9,") && text.contains(", r62, r63;"), "{name}: 64 registers"); + for k in 8..64 { + assert!(text.contains(&format!(" r{k} = r{} * 0x9e3779b9u + {k}u;\n", k & 7)), "{name}: r{k} derived"); + } + for k in 0..8 { + let terms: Vec = (k + 8..64).step_by(8).map(|j| format!("r{j}")).collect(); + assert!(text.contains(&format!(" r{k} = r{k} ^ {};\n", terms.join(" ^ "))), "{name}: the fold of r{k}"); + } + // a statement line opens with its register, or with the shuffle's block in OpenCL ("{ uint t_; ...") + let lines: Vec<&str> = text.lines().filter(|l| (l.starts_with(" r") || l.starts_with(" { uint t_;")) && l.contains(" // ")).collect(); + assert!(lines.len() % 128 == 0 && !lines.is_empty(), "{name}: {} statement lines, a multiple of 128", lines.len()); + for (k, (line, ins)) in lines.iter().take(128).zip(sched.iter()).enumerate() { + let stmt = line.trim_start(); + assert!(stmt.starts_with(&format!("r{} = ", ins.dst)) || (stmt.starts_with("{ uint t_;") && stmt.contains(&format!("r{} = r{} ^ t_", ins.dst, ins.dst))), "{name} statement {k}: dst r{}: {stmt}", ins.dst); + } + assert_eq!(text.matches("rotl_imm(").count() >= 2 * LOAD_SLOTS, true, "{name}: the era loads"); + } + // the pack: the vectors are the reg64 verifier's and differ from the plain pack's; program.h and program.json carry the flag + let out = export_pack(&e, &day_label("mx8-devnet-epoch0"), "test"); + for (i, &base) in out.bases.iter().enumerate() { + assert_eq!(out.outs[i], e.hash_warp(base)); + assert_ne!(out.outs[i], plain.hash_warp(base), "base {base}: the reg64 hashes differ from the plain pack's"); + } + let file = |n: &str| out.files.iter().find(|(f, _)| f == n).map(|(_, t)| t.clone()).unwrap(); + assert!(file("program.h").contains("#define IGNEUM_REG64 1\n#define IGNEUM_REGISTERS 64\n")); + let j: Value = serde_json::from_str(&file("program.json")).unwrap(); + assert_eq!(j["load_class"].as_str().unwrap(), "mx8-erad810f22d+reg64"); + assert_eq!(j["reg64"]["registers"].as_u64().unwrap(), 64); + assert_eq!(j["reg64"]["statements_per_iteration"].as_u64().unwrap(), 128); + assert_eq!(j["program_class"].as_str().unwrap(), "v3"); + let v: Value = serde_json::from_str(&file("vectors.json")).unwrap(); + assert_eq!(v["warps"].as_array().unwrap().len(), 3); +} + +/// The full-chain variant (`+reg64c`): every load's address source is `src ^ m` with `m` the rotate-xor mix of all +/// 64 registers; the CUDA text carries the mix statement before each of the 32 loads, the verifier the same chain, +/// and the hashes differ from the arithmetic-only window's. +#[test] +fn reg64_full_chain_cpu_verifier_and_cuda_text_agree() { + let mut e = epoch_reg64_of_devnet(); + let arithmetic_only: Vec<[u64; 32]> = [0u32, 4096].iter().map(|&b| e.hash_warp(b)).collect(); + e.program.class = e.program.class.with_reg64_chain(); + let p = &e.program; + assert!(p.address_mix()); + assert_eq!(p.class.name(), "mx8-erad810f22d+reg64c"); + assert_eq!(LoadClass::parse("mx8+reg64c"), Some(V3_CLASS.with_reg64().with_reg64_chain())); + assert_ne!(LoadClass::parse("mx8+reg64c"), LoadClass::parse("mx8+reg64")); + assert_ne!(p.program_id(), epoch_reg64_of_devnet().program.program_id(), "the id carries the chain"); + assert_ne!(p.program_id(), 0x3deee2320e70e1bf, "the sound fold's id differs from the benched text's (the source inside the chain)"); + assert_eq!(p.scheduled().len(), 128, "the schedule is the window's"); + let mp = e.dataset.memhard().map(|m| &m.params); + let sched = p.scheduled(); + let load_srcs: Vec = sched.iter().filter(|i| i.op == Op::Load).map(|i| i.src).collect(); + for (name, text) in [("kernel.cu", cuda_kernel(p, mp)), ("kernel_bound.cu", cuda_kernel_bound(p, mp))] { + assert!(text.contains(" uint32_t m; // reg64 full chain"), "{name}: m declared"); + assert_eq!(text.matches(" ^ m)").count(), 2 * LOAD_SLOTS, "{name}: 32 mixed address sources"); + assert_eq!(text.matches(" & mask]").count(), 2 * LOAD_SLOTS, "{name}: 32 masked loads"); + // the mix line sits right before its load: 63 terms in index order, the load's own source left out, and the + // load reads (rS ^ m) with the schedule's src + let lines: Vec<&str> = text.lines().collect(); + let mut loads = 0; + for (i, l) in lines.iter().enumerate() { + let t = l.trim_start(); + if t.starts_with("m = r") && t.contains("rotl_imm(m, 1u)") { + let src = load_srcs[loads]; + let next = lines[i + 1].trim_start(); + assert!(next.contains("ds[") && next.contains(&format!("(r{src} ^ m)")), "{name}: load {loads} follows its mix with src r{src}: {next}"); + let mut want = String::new(); + for k in 0..64u8 { + if k == src { + continue; + } + if want.is_empty() { + want.push_str(&format!("m = r{k};")); + } else { + want.push_str(&format!(" m = rotl_imm(m, 1u) ^ r{k};")); + } + } + assert_eq!(t, want, "{name}: the mix of load {loads} (src r{src})"); + loads += 1; + } + } + assert_eq!(loads, 2 * LOAD_SLOTS, "{name}: the mix before each of the 32 loads"); + } + for (i, &b) in [0u32, 4096].iter().enumerate() { + assert_ne!(e.hash_warp(b), arithmetic_only[i], "base {b}: the chain's hashes differ from the window's"); + } + let out = export_pack(&e, &day_label("mx8-devnet-epoch0"), "test"); + let file = |n: &str| out.files.iter().find(|(f, _)| f == n).map(|(_, t)| t.clone()).unwrap(); + assert!(file("program.h").contains("#define IGNEUM_REG64_ADDRESS_MIX 1")); + let j: Value = serde_json::from_str(&file("program.json")).unwrap(); + assert_eq!(j["load_class"].as_str().unwrap(), "mx8-erad810f22d+reg64c"); + assert_eq!(j["reg64"]["variant"].as_str().unwrap(), "window, full chain"); + assert_eq!(j["reg64"]["address_mix"].as_u64().unwrap(), 1); + assert!(j["reg64"]["liveness"].as_str().unwrap().contains("64 independently necessary values")); + assert_eq!(j["program_class"].as_str().unwrap(), "v3"); +} + +/// The window's liveness rule (`accept::check_window_liveness`): the arithmetic-only window is the known-failed +/// fixture (a load's address reads its own window's register, so complementing a register of another octet leaves +/// iteration 0's first load address where it was); the full-chain form passes; a program without the window is +/// `NotAWindow`. +#[test] +fn reg64_liveness_rule_refuses_the_subset_fold_and_passes_the_full_chain() { + use igneum_pow::accept::{check_window_liveness, Reject}; + let plain = epoch("mx8-devnet-epoch0"); + assert_eq!(check_window_liveness(&plain.program), Err(Reject::NotAWindow)); + let window = epoch_reg64_of_devnet(); + match check_window_liveness(&window.program) { + Err(Reject::DeadWindowRegister { address_changed, result_changed, .. }) => { + assert!(!address_changed, "the subset fold leaves the first load address where it was"); + assert!(result_changed, "the end fold still reads every register"); + } + other => panic!("the arithmetic-only window must be refused as a dead register: {other:?}"), + } + let mut chain = epoch_reg64_of_devnet(); + chain.program.class = chain.program.class.with_reg64_chain(); + assert_eq!(check_window_liveness(&chain.program), Ok(()), "the full chain keeps all 64 registers live across the chain"); + // wired into the acceptance (8 October 2026): `accept::check` refuses the arithmetic-only window as a dead register + // and never refuses the chain program for liveness; the plain pack's verdict does not move + use igneum_pow::accept::check; + assert!(matches!(check(&window.program), Err(Reject::DeadWindowRegister { .. })), "the acceptance refuses the arithmetic-only window"); + assert!(!matches!(check(&chain.program), Err(Reject::DeadWindowRegister { .. }) | Err(Reject::NotAWindow)), "the chain program is never refused for liveness"); + assert!(check(&plain.program).is_ok(), "the pinned devnet program's verdict does not move"); +} + +/// The external review's A07 (8 October 2026): the bound hash of a nonce is the lane of the aligned 32-nonce group that +/// contains it, in the verifier by construction (`Epoch::hash_bound`), across the group's tail and the low-32 rollover; +/// the high 32 bits of the nonce enter the init words, so two nonces 2^32 apart never share a hash. The launchers +/// refuse a job whose nonce_start is not 32-aligned or whose count is not a multiple of 32 (worker.cpp's job line). +#[test] +fn a07_bound_hash_is_group_aligned_across_tails_and_rollover() { + use igneum_pow::bind::block_init_words; + let e = epoch("mx8-devnet-epoch0"); + let prehash = [0x5au8; 32]; + for &n in &[0u64, 1, 2, 31, 32, 33, 63, (1u64 << 32) - 1, 1u64 << 32, (1u64 << 32) + 1, u64::MAX - 1, u64::MAX] { + let base = n & !31; + let warp = e.hash_warp_bound(&prehash, base); + assert_eq!(e.hash_bound(&prehash, n), warp[(n & 31) as usize], "nonce {n}: the lane of its aligned group"); + assert_eq!(e.hash_warp_bound(&prehash, n), warp, "nonce {n}: the group is the same from any of its nonces"); + let distinct: std::collections::HashSet = warp.iter().copied().collect(); + assert!(distinct.len() >= 31, "nonce {n}: the lanes of a group are distinct hashes"); + } + assert_ne!(e.hash_bound(&prehash, 2), e.hash_bound(&prehash, 2 + (1u64 << 32)), "the high 32 bits reach the hash"); + assert_ne!(block_init_words(&prehash, 2), block_init_words(&prehash, 2 + (1u64 << 32))); +} + +/// The external review's A04 (8 October 2026): the acceptance executes the shadow block as the hash does (class v4 +/// sub-version 3), so a shadow that zeroes the registers fails acceptance; the pinned program with its own shadow passes. +#[test] +fn a04_a_zeroing_shadow_fails_acceptance() { + use igneum_pow::accept::check; + // a class v4 program (the shadow block is class v4's): the same seed and era as the lib's program_classes test + let era = [7u8; 32]; + let p = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era)); + assert!(check(&p).is_ok(), "the class v4 program of the genesis seed passes"); + let mut z = p.clone(); + assert!(!z.shadow.is_empty(), "class v4 has a shadow block"); + for ins in z.shadow.iter_mut() { + ins.op = Op::Xor; + ins.src = ins.dst; + } + assert!(check(&z).is_err(), "a shadow of xor d, d zeroes every register it touches and must fail"); +} + +/// The external review's A06 (8 October 2026): the pack's identity.json authenticates every kernel text by BLAKE2b-256. +#[test] +fn a06_program_json_hashes_every_kernel_text() { + let e = epoch("mx8-devnet-epoch0"); + let out = export_pack(&e, &day_label("mx8-devnet-epoch0"), "test"); + let file = |n: &str| out.files.iter().find(|(f, _)| f == n).map(|(_, t)| t.clone()).unwrap(); + let j: Value = serde_json::from_str(&file("identity.json")).unwrap(); + let h = j["kernel_blake2b256"].as_object().expect("kernel_blake2b256"); + assert_eq!(j["program_id"].as_str().unwrap(), format!("{:#018x}", e.program.program_id())); + for n in ["kernel.cu", "kernel.cl", "program.metal", "program_bound.metal", "kernel_bound.cu", "kernel_bound.cl"] { + let want: String = igneum_pow::blake2b::blake2b_256(&[file(n).as_bytes()]).iter().map(|x| format!("{x:02x}")).collect(); + assert_eq!(h[n].as_str().unwrap(), want, "{n}"); + } +}