diff --git a/igneum-pow/src/emit.rs b/igneum-pow/src/emit.rs index 13ffe7d88..220db4306 100644 --- a/igneum-pow/src/emit.rs +++ b/igneum-pow/src/emit.rs @@ -1174,7 +1174,9 @@ fn reg_decl(p: &Program, ty: &str) -> String { } let names: Vec = (0..p.registers()).map(|k| format!("r{k}")).collect(); let m = if p.address_mix() { - if ty == "uint32_t" && reg64_prefix() { + if ty == "uint32_t" && reg64_form() == crate::reg64form::Reg64Form::Generated { + format!("\n {ty} m, S, p_, g0, g1, g2, g3, g4, g5, g6, g7; // reg64 full chain in the X4 generated form: the eight group sums, recomputed where the schedule wrote") + } else if ty == "uint32_t" && reg64_prefix() { format!("\n {ty} m, S, p_, t_; // reg64 full chain in the F05 prefix form: S the running xor of the rotated registers, p_ the prefix, t_ the old term") } else { format!("\n {ty} m; // reg64 full chain: the address mix of all 64 registers before every load") @@ -1194,14 +1196,65 @@ fn reg_decl(p: &Program, ty: &str) -> String { /// `r[s] ^ ror(P_s, 1) ^ (S ^ P_s ^ a[s])` with `P_s` the xor of the first `s` terms. The same hash, the same vectors /// (`verify::reg64_address_source`); a measurement text for the fleet, never the definition. OpenCL and Metal keep /// the fold. -static REG64_PREFIX: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); - +/// X4 of the 1.5x programme (9 October 2026): the switch is one value of the CUDA text's form +/// (`crate::reg64form`, `igneum-pow export --reg64-form fold|direct|generated|prefix`); `--reg64-prefix` is `prefix`. pub fn set_reg64_prefix(on: bool) { - REG64_PREFIX.store(on, std::sync::atomic::Ordering::Relaxed); + crate::reg64form::set_gpu_form(if on { crate::reg64form::Reg64Form::Prefix } else { crate::reg64form::Reg64Form::Fold }); } pub fn reg64_prefix() -> bool { - REG64_PREFIX.load(std::sync::atomic::Ordering::Relaxed) + crate::reg64form::gpu_form() == crate::reg64form::Reg64Form::Prefix +} + +/// The CUDA text's form of the reg64 address mix (the fold unless `--reg64-form` names another). +fn reg64_form() -> crate::reg64form::Reg64Form { + crate::reg64form::gpu_form() +} + +/// X4 direct form: the mix statement before a load of source `src` as the definition's closed sum, one rotation +/// per term (`rotl(rK, (62 - k) mod 32)` before the source, `(63 - k) mod 32` after it), xored as a balanced tree. +fn reg64_direct_mix_line(src: u8) -> String { + let s = src as usize; + let terms: Vec = (0..64) + .filter(|&k| k != s) + .map(|k| { + let n = crate::reg64form::direct_rot(k, s); + if n == 0 { format!("r{k}") } else { format!("rotl_imm(r{k}, {n}u)") } + }) + .collect(); + format!("m = {};", xor_tree(&terms)) +} + +/// A balanced xor tree of `terms` as text (parenthesised so the compiler is handed the tree, not a chain). +fn xor_tree(terms: &[String]) -> String { + match terms.len() { + 0 => "0u".to_string(), + 1 => terms[0].clone(), + n => format!("({} ^ {})", xor_tree(&terms[..n / 2]), xor_tree(&terms[n / 2..])), + } +} + +/// X4 generated form: one group sum `gJ = a[8J] ^ ... ^ a[8J + 7]` as a statement. +fn reg64_group_line(j: usize) -> String { + let terms: Vec = (j * 8..j * 8 + 8).map(reg64_term).collect(); + format!("g{j} = {};", xor_tree(&terms)) +} + +/// X4 generated form: the statements before a load of source `src` (the plan's stale groups recomputed, then +/// `m = ror(P, 1) ^ S ^ P ^ a[src]` with `S` the eight groups and `P` the groups before src's plus the partial). +fn reg64_generated_mix_line(recompute: &[usize], src: u8) -> String { + let s = src as usize; + let mut out = String::new(); + for &j in recompute { + out.push_str(®64_group_line(j)); + out.push(' '); + } + let groups: Vec = (0..8).map(|j| format!("g{j}")).collect(); + let jj = s / 8; + let mut prefix: Vec = (0..jj).map(|j| format!("g{j}")).collect(); + prefix.extend((jj * 8..s).map(reg64_term)); + out.push_str(&format!("S = {}; p_ = {}; m = rotr_var(p_, 1u) ^ S ^ p_ ^ {};", xor_tree(&groups), xor_tree(&prefix), reg64_term(s))); + out } /// `rotl(rK, (63 - k) mod 32)` as text; a rotation of 0 is the register itself (`rotl_imm` takes 1..31). @@ -1220,6 +1273,13 @@ fn reg64_prefix_mix_line(src: u8) -> String { /// The prefix form's running total after the register init: `S = a[0] ^ ... ^ a[63]`. fn reg64_prefix_init(p: &Program) -> String { + if p.class.reg64 && p.address_mix() && reg64_form() == crate::reg64form::Reg64Form::Generated { + let mut s = String::from(" // X4 generated form: the eight group sums of the rotated registers\n"); + for j in 0..8 { + s.push_str(&format!(" {}\n", reg64_group_line(j))); + } + return s; + } if !p.class.reg64 || !p.address_mix() || !reg64_prefix() { return String::new(); } @@ -1278,13 +1338,22 @@ fn cuda_instr_lines(p: &Program, geom: DatasetGeom) -> String { // just before the load (the verifier's addr_src, the same chain) let address_mix = p.address_mix(); let addr_src = |a: &str| -> String { if address_mix { format!("({a} ^ m)") } else { a.to_string() } }; - for (k, ins) in p.scheduled().iter().enumerate() { + let scheduled = p.scheduled(); + let form = reg64_form(); + let plan = if address_mix && form == crate::reg64form::Reg64Form::Generated { crate::reg64form::generated_plan(&scheduled, &p.shadow) } else { Vec::new() }; + for (k, ins) in scheduled.iter().enumerate() { let d = format!("r{}", ins.dst); let a = format!("r{}", ins.src); let b = format!("r{}", ins.src2); let prefix = address_mix && reg64_prefix(); if address_mix && ins.op == Op::Load { - s.push_str(&format!(" {}\n", if prefix { reg64_prefix_mix_line(ins.src) } else { reg64_mix_line(p, ins.src) })); + let line = match form { + crate::reg64form::Reg64Form::Prefix => reg64_prefix_mix_line(ins.src), + crate::reg64form::Reg64Form::Direct => reg64_direct_mix_line(ins.src), + crate::reg64form::Reg64Form::Generated => reg64_generated_mix_line(&plan[k], ins.src), + _ => reg64_mix_line(p, ins.src), + }; + s.push_str(&format!(" {line}\n")); } if prefix { // the old term of the destination, so S can drop it after the write diff --git a/igneum-pow/src/lib.rs b/igneum-pow/src/lib.rs index 690ddb0bc..32aa9d7d0 100644 --- a/igneum-pow/src/lib.rs +++ b/igneum-pow/src/lib.rs @@ -30,6 +30,7 @@ pub mod emit; pub mod generator; pub mod memhard; pub mod packcheck; +pub mod reg64form; pub mod seed; pub mod state; pub mod verify; diff --git a/igneum-pow/src/main.rs b/igneum-pow/src/main.rs index 205f6ce66..fc81d0955 100644 --- a/igneum-pow/src/main.rs +++ b/igneum-pow/src/main.rs @@ -123,7 +123,9 @@ fn usage() -> ! { \x20 --dataset-words N research class ds55: the dataset at N words (a multiple of 65,536 in 2^28 ..= 2^31; 1476395008 = 5.5 GiB); a non-power-of-two uses idx = (src * N) >> 32 in every load (spec 01 section 1.13.3), a power of two is --dataset-log2\n\\ \x20 --reg64 the 64-register window per lane over the class (research; the same as --class +reg64; CUDA, OpenCL and Metal texts)\n\ \x20 --reg64-chain reg64 with the full-chain address mix: every load's address consumes all 64 registers (the same as --class +reg64c)\n\ - \x20 --reg64-prefix export: the CUDA texts carry the reg64 chain in its closed prefix form (review B's F05; the same hash and vectors, a measurement text)" + \x20 --reg64-prefix export: the CUDA texts carry the reg64 chain in its closed prefix form (review B's F05; the same hash and vectors, a measurement text)\n\ + \x20 --reg64-form export: the CUDA texts' reg64 address-mix form, fold (the default), direct, generated or prefix (X4, 1.5x programme; the same hash)\n\ + \x20 --reg64-cpu-form the CPU interpreter computes the reg64 address mix in form f (fold, direct, generated, prefix, or wrong: the known-failed form), the X4 differential only" ); std::process::exit(2) } @@ -191,6 +193,11 @@ fn parse() -> Args { a.reg64_chain = true; } "--reg64-prefix" => igneum_pow::emit::set_reg64_prefix(true), + "--reg64-form" => match igneum_pow::reg64form::Reg64Form::parse(&val()) { + Some(igneum_pow::reg64form::Reg64Form::Wrong) | None => usage(), + Some(f) => igneum_pow::reg64form::set_gpu_form(f), + }, + "--reg64-cpu-form" => igneum_pow::reg64form::set_cpu_form(igneum_pow::reg64form::Reg64Form::parse(&val()).unwrap_or_else(|| usage())), _ => usage(), } } diff --git a/igneum-pow/src/reg64form.rs b/igneum-pow/src/reg64form.rs new file mode 100644 index 000000000..d0de3dbfb --- /dev/null +++ b/igneum-pow/src/reg64form.rs @@ -0,0 +1,322 @@ +//! X4 of the 1.5x programme (9 October 2026, the k lane): byte-identical implementation forms of the class v6 +//! reg64c address mix (`docs/analysis/class-v6/1p5x/x4/README.md`). The DEFINITION stays the verifier's fold +//! (`verify::step`'s `addr_src`: `m = r[k0]; m = rotl(m, 1) ^ r[k1]; ...` over the 63 registers other than the load's +//! source `s`, in index order, the address source `r[s] ^ m`). Every form here computes the same 32-bit word on every +//! register state and every source; none of them is a consensus change, and the default everywhere is the fold. +//! +//! * `Fold`: the frozen text (the 63-term dependent chain per load). +//! * `Direct`: the definition's closed sum written directly, `m = XOR_{ks} rotl(r[k], (63 - k) mod 32)`, a balanced XOR tree with no carried chain and no state. +//! * `Generated`: a code-generated form over eight group sums `g_j = XOR_{k in 8j..8j+7} a[k]`, `a[k] = rotl(r[k], +//! (63 - k) mod 32)`. The generator reads the static schedule and recomputes, before each load, only the groups +//! written since the previous load ([`generated_plan`]); the shadow block draws its registers from r0..r7 +//! (`generator.rs`, `dst = rng.below(8)`), so the 55,296 shadow writes dirty group 0 only. A load's source is +//! `r[s] ^ ror(P_s, 1) ^ S ^ P_s ^ a[s]` with `S = XOR g_j` and `P_s` the whole groups before `s`'s group plus the +//! in-group partial (review B's F05 identity, `verify::reg64_address_source`). +//! * `Prefix`: review B's F05 text as landed (`--reg64-prefix`): a running `S` updated after every write (main and +//! shadow), `P_s` recomputed per load. +//! * `Wrong`: the known-failed variant for the differential test only (`r[s] ^ S ^ a[s]`, every term rotated as if it +//! sat after the source); the differential must refuse it. Never emitted. + +use std::sync::atomic::{AtomicU8, Ordering}; + +use crate::generator::{Instr, Op, LANES}; + +/// Registers per group of the generated form. +pub const GROUP: usize = 8; +/// Groups of the generated form (64 registers). +pub const GROUPS: usize = 8; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Reg64Form { + Fold, + Direct, + Generated, + Prefix, + Wrong, +} + +impl Reg64Form { + pub fn parse(s: &str) -> Option { + Some(match s { + "fold" => Reg64Form::Fold, + "direct" => Reg64Form::Direct, + "generated" => Reg64Form::Generated, + "prefix" => Reg64Form::Prefix, + "wrong" => Reg64Form::Wrong, + _ => return None, + }) + } + pub fn name(self) -> &'static str { + match self { + Reg64Form::Fold => "fold", + Reg64Form::Direct => "direct", + Reg64Form::Generated => "generated", + Reg64Form::Prefix => "prefix", + Reg64Form::Wrong => "wrong", + } + } + fn code(self) -> u8 { + self as u8 + } + fn from_code(c: u8) -> Reg64Form { + [Reg64Form::Fold, Reg64Form::Direct, Reg64Form::Generated, Reg64Form::Prefix, Reg64Form::Wrong][c as usize] + } +} + +static GPU_FORM: AtomicU8 = AtomicU8::new(0); +static CPU_FORM: AtomicU8 = AtomicU8::new(0); + +/// The CUDA text's form (`igneum-pow export --reg64-form
`; `--reg64-prefix` is `prefix`). `Wrong` is refused. +pub fn set_gpu_form(f: Reg64Form) { + assert!(f != Reg64Form::Wrong, "the known-failed form is never emitted"); + GPU_FORM.store(f.code(), Ordering::Relaxed); +} +pub fn gpu_form() -> Reg64Form { + Reg64Form::from_code(GPU_FORM.load(Ordering::Relaxed)) +} +/// The CPU interpreter's form (`--reg64-cpu-form `, the differential test's switch). The fold is the +/// definition and the default; every other form runs beside it only in the X4 differential. +pub fn set_cpu_form(f: Reg64Form) { + CPU_FORM.store(f.code(), Ordering::Relaxed); +} +pub fn cpu_form() -> Reg64Form { + Reg64Form::from_code(CPU_FORM.load(Ordering::Relaxed)) +} + +/// `a[k] = rotl(r[k], (63 - k) mod 32)`: register k's rotation in the full 64-register chain. +#[inline(always)] +pub fn term(r: &[u32; 64], k: usize) -> u32 { + r[k].rotate_left(((63 - k) % 32) as u32) +} + +/// The rotation register `k` carries in the chain of a load whose source is `s` (k != s). +#[inline(always)] +pub fn direct_rot(k: usize, s: usize) -> u32 { + if k < s { ((62 - k) % 32) as u32 } else { ((63 - k) % 32) as u32 } +} + +/// The definition, one lane: `r[s] ^ m` with the 63-term chain (the same loop as `verify::step`). +pub fn fold(r: &[u32; 64], s: usize) -> u32 { + let mut m = 0u32; + let mut started = false; + for (k, &v) in r.iter().enumerate() { + if k == s { + continue; + } + m = if started { m.rotate_left(1) ^ v } else { v }; + started = true; + } + r[s] ^ m +} + +/// The direct form: the closed sum, one rotation per term. +pub fn direct(r: &[u32; 64], s: usize) -> u32 { + let mut m = 0u32; + for k in 0..64 { + if k != s { + m ^= r[k].rotate_left(direct_rot(k, s)); + } + } + r[s] ^ m +} + +/// One group sum of the generated form. +#[inline(always)] +pub fn group(r: &[u32; 64], j: usize) -> u32 { + let mut g = 0u32; + for k in j * GROUP..(j + 1) * GROUP { + g ^= term(r, k); + } + g +} + +/// The generated form's query from the cached group sums `g` (stale groups give a wrong word: the plan's job). +pub fn grouped(g: &[u32; GROUPS], r: &[u32; 64], s: usize) -> u32 { + let total = g.iter().fold(0u32, |x, y| x ^ y); + let j = s / GROUP; + let mut p = g[..j].iter().fold(0u32, |x, y| x ^ y); + for k in j * GROUP..s { + p ^= term(r, k); + } + r[s] ^ p.rotate_right(1) ^ total ^ p ^ term(r, s) +} + +/// The prefix form's query from a running total `total` (= XOR of every `a[k]` when kept fresh). +pub fn prefixed(total: u32, r: &[u32; 64], s: usize) -> u32 { + let mut p = 0u32; + for k in 0..s { + p ^= term(r, k); + } + r[s] ^ p.rotate_right(1) ^ total ^ p ^ term(r, s) +} + +/// The known-failed form (the F05 test's wrong variant). +pub fn wrong(r: &[u32; 64], s: usize) -> u32 { + let total = (0..64).map(|k| term(r, k)).fold(0, |x, y| x ^ y); + r[s] ^ total ^ term(r, s) +} + +/// The generated form's static plan: for each statement of the scheduled list, the groups to recompute just before +/// it (non-empty only at loads). The groups entering an iteration are those written after the last load of the +/// previous iteration's list plus every group the shadow block writes (group 0 on the shipped class); the kernel +/// computes all eight after the register init, so iteration 0 recomputes a superset (idempotent). +pub fn generated_plan(scheduled: &[Instr], shadow: &[Instr]) -> Vec> { + let all: u8 = 0xff; + let bit = |ins: &Instr| 1u8 << (ins.dst as usize / GROUP); + let mut dirty = all; + for ins in scheduled { + if ins.op == Op::Load { + dirty = 0; + } + dirty |= bit(ins); + } + let shadow_groups = shadow.iter().fold(0u8, |m, ins| m | bit(ins)); + let mut dirty = dirty | shadow_groups; + let mut plan = Vec::with_capacity(scheduled.len()); + for ins in scheduled { + if ins.op == Op::Load { + plan.push((0..GROUPS).filter(|j| dirty & (1 << j) != 0).collect()); + dirty = 0; + } else { + plan.push(Vec::new()); + } + dirty |= bit(ins); + } + plan +} + +/// The groups the shadow block writes (a bit per group). +pub fn shadow_group_mask(shadow: &[Instr]) -> u8 { + shadow.iter().fold(0u8, |m, ins| m | (1u8 << (ins.dst as usize / GROUP))) +} + +/// Counts of the generated plan per iteration: (loads, group recomputes) — the op count's input. +pub fn plan_counts(plan: &[Vec], scheduled: &[Instr]) -> (usize, usize) { + let loads = scheduled.iter().filter(|i| i.op == Op::Load).count(); + let recomputes = plan.iter().map(|v| v.len()).sum(); + (loads, recomputes) +} + +/// The CPU interpreter's state for a form (the differential's emulation of each kernel text): the generated +/// form's eight cached group sums per lane, recomputed only where [`generated_plan`] says; the prefix form's running +/// total per lane, updated after every write (main list and shadow), as the F05 text does. +pub struct FormState { + form: Reg64Form, + plan: Vec>, + groups: [[u32; LANES]; GROUPS], + total: [u32; LANES], +} + +fn lane_regs(r: &[[u32; LANES]], lane: usize) -> [u32; 64] { + std::array::from_fn(|k| r[k][lane]) +} + +impl FormState { + pub fn new(form: Reg64Form, plan: Vec>) -> FormState { + FormState { form, plan, groups: [[0; LANES]; GROUPS], total: [0; LANES] } + } + /// After the register init (and the probe's flip): every group and the running total from the live registers. + pub fn init(&mut self, r: &[[u32; LANES]]) { + for lane in 0..LANES { + let regs = lane_regs(r, lane); + for j in 0..GROUPS { + self.groups[j][lane] = group(®s, j); + } + self.total[lane] = (0..64).map(|k| term(®s, k)).fold(0, |x, y| x ^ y); + } + } + /// After a write of register `d` whose value before the write was `old`: the prefix form's running total moves + /// by the term's change; the generated form keeps its groups stale until the plan recomputes them. + pub fn after_write(&mut self, r: &[[u32; LANES]], d: usize, old: &[u32; LANES]) { + if self.form == Reg64Form::Prefix { + let rot = ((63 - d) % 32) as u32; + for lane in 0..LANES { + self.total[lane] ^= old[lane].rotate_left(rot) ^ r[d][lane].rotate_left(rot); + } + } + } + /// The address source of the load at scheduled position `k` with source `s`, per lane, in this form. + pub fn source(&mut self, r: &[[u32; LANES]], k: usize, s: usize) -> [u32; LANES] { + if self.form == Reg64Form::Generated { + for &j in &self.plan[k] { + for lane in 0..LANES { + let regs = lane_regs(r, lane); + self.groups[j][lane] = group(®s, j); + } + } + } + std::array::from_fn(|lane| { + let regs = lane_regs(r, lane); + match self.form { + Reg64Form::Fold => fold(®s, s), + Reg64Form::Direct => direct(®s, s), + Reg64Form::Generated => grouped(&std::array::from_fn(|j| self.groups[j][lane]), ®s, s), + Reg64Form::Prefix => prefixed(self.total[lane], ®s, s), + Reg64Form::Wrong => wrong(®s, s), + } + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn stream(seed: u64) -> impl FnMut() -> u32 { + let mut x = seed; + move || { + x = x.wrapping_add(0x9e37_79b9_7f4a_7c15); + let mut z = x; + z = (z ^ (z >> 30)).wrapping_mul(0xbf58_476d_1ce4_e5b9); + z = (z ^ (z >> 27)).wrapping_mul(0x94d0_49bb_1331_11eb); + (z ^ (z >> 31)) as u32 + } + } + + /// Known-failed first: the wrong form must disagree with the fold somewhere; then the three forms agree with the + /// fold on every source over 4,096 random states and on the structured states (all zero, all ones, one bit, + /// the init rule's registers). + #[test] + fn x4_forms_equal_the_fold_and_the_wrong_form_is_refused() { + let mut next = stream(0x1505_0004); + let mut wrong_differs = 0usize; + let mut states: Vec<[u32; 64]> = Vec::new(); + for _ in 0..4096 { + let mut r = [0u32; 64]; + for v in r.iter_mut() { + *v = next(); + } + states.push(r); + } + states.push([0u32; 64]); + states.push([u32::MAX; 64]); + for b in 0..64 * 32 { + let mut r = [0u32; 64]; + r[b / 32] = 1 << (b % 32); + states.push(r); + } + let mut r = [0u32; 64]; + for k in 0..8 { + r[k] = next(); + } + for k in 8..64 { + r[k] = r[k & 7].wrapping_mul(0x9e37_79b9).wrapping_add(k as u32); + } + states.push(r); + for r in &states { + let g: [u32; GROUPS] = std::array::from_fn(|j| group(r, j)); + let total = (0..64).map(|k| term(r, k)).fold(0, |x, y| x ^ y); + for s in 0..64 { + let want = fold(r, s); + if wrong(r, s) != want { + wrong_differs += 1; + } + assert_eq!(direct(r, s), want, "direct, source r{s}"); + assert_eq!(grouped(&g, r, s), want, "generated, source r{s}"); + assert_eq!(prefixed(total, r, s), want, "prefix, source r{s}"); + } + } + assert!(wrong_differs > 0, "the known-failed form must be refused"); + } +} diff --git a/igneum-pow/src/verify.rs b/igneum-pow/src/verify.rs index c0dc217b8..895a409ec 100644 --- a/igneum-pow/src/verify.rs +++ b/igneum-pow/src/verify.rs @@ -581,6 +581,10 @@ fn interpret_warp_core( // the iteration's statements: the drawn program, or its interleaved two-window form under reg64 let scheduled = program.scheduled(); let address_mix = program.address_mix(); + // X4: a byte-identical form of the address mix beside the definition (crate::reg64form), off by default + let cpu_form = crate::reg64form::cpu_form(); + let form_on = address_mix && cpu_form != crate::reg64form::Reg64Form::Fold; + let mut fs = crate::reg64form::FormState::new(cpu_form, if form_on { crate::reg64form::generated_plan(&scheduled, &program.shadow) } else { Vec::new() }); let mut items_derived = 0usize; let mut idx = [0u32; LANES]; let mut val = [0u32; LANES]; @@ -606,9 +610,15 @@ fn interpret_warp_core( } } } + if it == 0 && form_on { + // X4: the form's state from the registers as iteration 0 sees them (after the init and any probe flip) + fs.init(&r); + } let sel = r[0]; - for ins in &scheduled { - step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix); + for (k, ins) in scheduled.iter().enumerate() { + let old = if form_on { r[ins.dst as usize] } else { [0u32; LANES] }; + let ovr = if form_on && ins.op == Op::Load { Some(fs.source(&r, k, ins.src as usize)) } else { None }; + step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix, ovr.as_ref()); if it == 0 && ins.op == Op::Load { if let Some(pr) = probe.as_deref_mut() { if !pr.seen_first_load { @@ -625,12 +635,19 @@ fn interpret_warp_core( r[d][lane] = m.rmw(&program.seed, base_nonce, lane, slot, r[d][lane]); } } + if form_on { + fs.after_write(&r, ins.dst as usize, &old); + } } // Latency-shadow block (Counter ASIC 3.0 item 8): the block runs `reps` times after instruction 63 with the // iteration's `sel`; it is empty on every class without a shadow, so version 2 and class v3 run nothing here. for _ in 0..program.shadow_reps() { for ins in &program.shadow { - step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix); + let old = if form_on { r[ins.dst as usize] } else { [0u32; LANES] }; + step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix, None); + if form_on { + fs.after_write(&r, ins.dst as usize, &old); + } } } } @@ -668,6 +685,7 @@ fn step( val: &mut [u32; LANES], items_derived: &mut usize, address_mix: bool, + src_override: Option<&[u32; LANES]>, ) { let d = ins.dst as usize; let a = ins.src as usize; @@ -679,6 +697,11 @@ fn step( if !address_mix { return r[a][lane]; } + // X4 (1.5x programme): a byte-identical form computed by the caller; never set unless the differential's + // `--reg64-cpu-form` names a form other than the fold (crate::reg64form) + if let Some(o) = src_override { + return o[lane]; + } let mut m = 0u32; let mut started = false; for k in 0..r.len() {