X4 (1.5x): byte-identical reg64c address-mix forms (direct, generated, prefix) behind --reg64-form and --reg64-cpu-form; the fold stays the definition and the default
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
688904914f
commit
e1a910c7fc
5 changed files with 433 additions and 11 deletions
|
|
@ -1174,7 +1174,9 @@ fn reg_decl(p: &Program, ty: &str) -> String {
|
|||
}
|
||||
let names: Vec<String> = (0..p.registers()).map(|k| format!("r{k}")).collect();
|
||||
let m = if p.address_mix() {
|
||||
if ty == "uint32_t" && reg64_prefix() {
|
||||
if ty == "uint32_t" && reg64_form() == crate::reg64form::Reg64Form::Generated {
|
||||
format!("\n {ty} m, S, p_, g0, g1, g2, g3, g4, g5, g6, g7; // reg64 full chain in the X4 generated form: the eight group sums, recomputed where the schedule wrote")
|
||||
} else if ty == "uint32_t" && reg64_prefix() {
|
||||
format!("\n {ty} m, S, p_, t_; // reg64 full chain in the F05 prefix form: S the running xor of the rotated registers, p_ the prefix, t_ the old term")
|
||||
} else {
|
||||
format!("\n {ty} m; // reg64 full chain: the address mix of all 64 registers before every load")
|
||||
|
|
@ -1194,14 +1196,65 @@ fn reg_decl(p: &Program, ty: &str) -> String {
|
|||
/// `r[s] ^ ror(P_s, 1) ^ (S ^ P_s ^ a[s])` with `P_s` the xor of the first `s` terms. The same hash, the same vectors
|
||||
/// (`verify::reg64_address_source`); a measurement text for the fleet, never the definition. OpenCL and Metal keep
|
||||
/// the fold.
|
||||
static REG64_PREFIX: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false);
|
||||
|
||||
/// X4 of the 1.5x programme (9 October 2026): the switch is one value of the CUDA text's form
|
||||
/// (`crate::reg64form`, `igneum-pow export --reg64-form fold|direct|generated|prefix`); `--reg64-prefix` is `prefix`.
|
||||
pub fn set_reg64_prefix(on: bool) {
|
||||
REG64_PREFIX.store(on, std::sync::atomic::Ordering::Relaxed);
|
||||
crate::reg64form::set_gpu_form(if on { crate::reg64form::Reg64Form::Prefix } else { crate::reg64form::Reg64Form::Fold });
|
||||
}
|
||||
|
||||
pub fn reg64_prefix() -> bool {
|
||||
REG64_PREFIX.load(std::sync::atomic::Ordering::Relaxed)
|
||||
crate::reg64form::gpu_form() == crate::reg64form::Reg64Form::Prefix
|
||||
}
|
||||
|
||||
/// The CUDA text's form of the reg64 address mix (the fold unless `--reg64-form` names another).
|
||||
fn reg64_form() -> crate::reg64form::Reg64Form {
|
||||
crate::reg64form::gpu_form()
|
||||
}
|
||||
|
||||
/// X4 direct form: the mix statement before a load of source `src` as the definition's closed sum, one rotation
|
||||
/// per term (`rotl(rK, (62 - k) mod 32)` before the source, `(63 - k) mod 32` after it), xored as a balanced tree.
|
||||
fn reg64_direct_mix_line(src: u8) -> String {
|
||||
let s = src as usize;
|
||||
let terms: Vec<String> = (0..64)
|
||||
.filter(|&k| k != s)
|
||||
.map(|k| {
|
||||
let n = crate::reg64form::direct_rot(k, s);
|
||||
if n == 0 { format!("r{k}") } else { format!("rotl_imm(r{k}, {n}u)") }
|
||||
})
|
||||
.collect();
|
||||
format!("m = {};", xor_tree(&terms))
|
||||
}
|
||||
|
||||
/// A balanced xor tree of `terms` as text (parenthesised so the compiler is handed the tree, not a chain).
|
||||
fn xor_tree(terms: &[String]) -> String {
|
||||
match terms.len() {
|
||||
0 => "0u".to_string(),
|
||||
1 => terms[0].clone(),
|
||||
n => format!("({} ^ {})", xor_tree(&terms[..n / 2]), xor_tree(&terms[n / 2..])),
|
||||
}
|
||||
}
|
||||
|
||||
/// X4 generated form: one group sum `gJ = a[8J] ^ ... ^ a[8J + 7]` as a statement.
|
||||
fn reg64_group_line(j: usize) -> String {
|
||||
let terms: Vec<String> = (j * 8..j * 8 + 8).map(reg64_term).collect();
|
||||
format!("g{j} = {};", xor_tree(&terms))
|
||||
}
|
||||
|
||||
/// X4 generated form: the statements before a load of source `src` (the plan's stale groups recomputed, then
|
||||
/// `m = ror(P, 1) ^ S ^ P ^ a[src]` with `S` the eight groups and `P` the groups before src's plus the partial).
|
||||
fn reg64_generated_mix_line(recompute: &[usize], src: u8) -> String {
|
||||
let s = src as usize;
|
||||
let mut out = String::new();
|
||||
for &j in recompute {
|
||||
out.push_str(®64_group_line(j));
|
||||
out.push(' ');
|
||||
}
|
||||
let groups: Vec<String> = (0..8).map(|j| format!("g{j}")).collect();
|
||||
let jj = s / 8;
|
||||
let mut prefix: Vec<String> = (0..jj).map(|j| format!("g{j}")).collect();
|
||||
prefix.extend((jj * 8..s).map(reg64_term));
|
||||
out.push_str(&format!("S = {}; p_ = {}; m = rotr_var(p_, 1u) ^ S ^ p_ ^ {};", xor_tree(&groups), xor_tree(&prefix), reg64_term(s)));
|
||||
out
|
||||
}
|
||||
|
||||
/// `rotl(rK, (63 - k) mod 32)` as text; a rotation of 0 is the register itself (`rotl_imm` takes 1..31).
|
||||
|
|
@ -1220,6 +1273,13 @@ fn reg64_prefix_mix_line(src: u8) -> String {
|
|||
|
||||
/// The prefix form's running total after the register init: `S = a[0] ^ ... ^ a[63]`.
|
||||
fn reg64_prefix_init(p: &Program) -> String {
|
||||
if p.class.reg64 && p.address_mix() && reg64_form() == crate::reg64form::Reg64Form::Generated {
|
||||
let mut s = String::from(" // X4 generated form: the eight group sums of the rotated registers\n");
|
||||
for j in 0..8 {
|
||||
s.push_str(&format!(" {}\n", reg64_group_line(j)));
|
||||
}
|
||||
return s;
|
||||
}
|
||||
if !p.class.reg64 || !p.address_mix() || !reg64_prefix() {
|
||||
return String::new();
|
||||
}
|
||||
|
|
@ -1278,13 +1338,22 @@ fn cuda_instr_lines(p: &Program, geom: DatasetGeom) -> String {
|
|||
// just before the load (the verifier's addr_src, the same chain)
|
||||
let address_mix = p.address_mix();
|
||||
let addr_src = |a: &str| -> String { if address_mix { format!("({a} ^ m)") } else { a.to_string() } };
|
||||
for (k, ins) in p.scheduled().iter().enumerate() {
|
||||
let scheduled = p.scheduled();
|
||||
let form = reg64_form();
|
||||
let plan = if address_mix && form == crate::reg64form::Reg64Form::Generated { crate::reg64form::generated_plan(&scheduled, &p.shadow) } else { Vec::new() };
|
||||
for (k, ins) in scheduled.iter().enumerate() {
|
||||
let d = format!("r{}", ins.dst);
|
||||
let a = format!("r{}", ins.src);
|
||||
let b = format!("r{}", ins.src2);
|
||||
let prefix = address_mix && reg64_prefix();
|
||||
if address_mix && ins.op == Op::Load {
|
||||
s.push_str(&format!(" {}\n", if prefix { reg64_prefix_mix_line(ins.src) } else { reg64_mix_line(p, ins.src) }));
|
||||
let line = match form {
|
||||
crate::reg64form::Reg64Form::Prefix => reg64_prefix_mix_line(ins.src),
|
||||
crate::reg64form::Reg64Form::Direct => reg64_direct_mix_line(ins.src),
|
||||
crate::reg64form::Reg64Form::Generated => reg64_generated_mix_line(&plan[k], ins.src),
|
||||
_ => reg64_mix_line(p, ins.src),
|
||||
};
|
||||
s.push_str(&format!(" {line}\n"));
|
||||
}
|
||||
if prefix {
|
||||
// the old term of the destination, so S can drop it after the write
|
||||
|
|
|
|||
|
|
@ -30,6 +30,7 @@ pub mod emit;
|
|||
pub mod generator;
|
||||
pub mod memhard;
|
||||
pub mod packcheck;
|
||||
pub mod reg64form;
|
||||
pub mod seed;
|
||||
pub mod state;
|
||||
pub mod verify;
|
||||
|
|
|
|||
|
|
@ -123,7 +123,9 @@ fn usage() -> ! {
|
|||
\x20 --dataset-words N research class ds55: the dataset at N words (a multiple of 65,536 in 2^28 ..= 2^31; 1476395008 = 5.5 GiB); a non-power-of-two uses idx = (src * N) >> 32 in every load (spec 01 section 1.13.3), a power of two is --dataset-log2\n\\
|
||||
\x20 --reg64 the 64-register window per lane over the class (research; the same as --class <class>+reg64; CUDA, OpenCL and Metal texts)\n\
|
||||
\x20 --reg64-chain reg64 with the full-chain address mix: every load's address consumes all 64 registers (the same as --class <class>+reg64c)\n\
|
||||
\x20 --reg64-prefix export: the CUDA texts carry the reg64 chain in its closed prefix form (review B's F05; the same hash and vectors, a measurement text)"
|
||||
\x20 --reg64-prefix export: the CUDA texts carry the reg64 chain in its closed prefix form (review B's F05; the same hash and vectors, a measurement text)\n\
|
||||
\x20 --reg64-form <f> export: the CUDA texts' reg64 address-mix form, fold (the default), direct, generated or prefix (X4, 1.5x programme; the same hash)\n\
|
||||
\x20 --reg64-cpu-form <f> the CPU interpreter computes the reg64 address mix in form f (fold, direct, generated, prefix, or wrong: the known-failed form), the X4 differential only"
|
||||
);
|
||||
std::process::exit(2)
|
||||
}
|
||||
|
|
@ -191,6 +193,11 @@ fn parse() -> Args {
|
|||
a.reg64_chain = true;
|
||||
}
|
||||
"--reg64-prefix" => igneum_pow::emit::set_reg64_prefix(true),
|
||||
"--reg64-form" => match igneum_pow::reg64form::Reg64Form::parse(&val()) {
|
||||
Some(igneum_pow::reg64form::Reg64Form::Wrong) | None => usage(),
|
||||
Some(f) => igneum_pow::reg64form::set_gpu_form(f),
|
||||
},
|
||||
"--reg64-cpu-form" => igneum_pow::reg64form::set_cpu_form(igneum_pow::reg64form::Reg64Form::parse(&val()).unwrap_or_else(|| usage())),
|
||||
_ => usage(),
|
||||
}
|
||||
}
|
||||
|
|
|
|||
322
igneum-pow/src/reg64form.rs
Normal file
322
igneum-pow/src/reg64form.rs
Normal file
|
|
@ -0,0 +1,322 @@
|
|||
//! X4 of the 1.5x programme (9 October 2026, the k lane): byte-identical implementation forms of the class v6
|
||||
//! reg64c address mix (`docs/analysis/class-v6/1p5x/x4/README.md`). The DEFINITION stays the verifier's fold
|
||||
//! (`verify::step`'s `addr_src`: `m = r[k0]; m = rotl(m, 1) ^ r[k1]; ...` over the 63 registers other than the load's
|
||||
//! source `s`, in index order, the address source `r[s] ^ m`). Every form here computes the same 32-bit word on every
|
||||
//! register state and every source; none of them is a consensus change, and the default everywhere is the fold.
|
||||
//!
|
||||
//! * `Fold`: the frozen text (the 63-term dependent chain per load).
|
||||
//! * `Direct`: the definition's closed sum written directly, `m = XOR_{k<s} rotl(r[k], (62 - k) mod 32) ^
|
||||
//! XOR_{k>s} rotl(r[k], (63 - k) mod 32)`, a balanced XOR tree with no carried chain and no state.
|
||||
//! * `Generated`: a code-generated form over eight group sums `g_j = XOR_{k in 8j..8j+7} a[k]`, `a[k] = rotl(r[k],
|
||||
//! (63 - k) mod 32)`. The generator reads the static schedule and recomputes, before each load, only the groups
|
||||
//! written since the previous load ([`generated_plan`]); the shadow block draws its registers from r0..r7
|
||||
//! (`generator.rs`, `dst = rng.below(8)`), so the 55,296 shadow writes dirty group 0 only. A load's source is
|
||||
//! `r[s] ^ ror(P_s, 1) ^ S ^ P_s ^ a[s]` with `S = XOR g_j` and `P_s` the whole groups before `s`'s group plus the
|
||||
//! in-group partial (review B's F05 identity, `verify::reg64_address_source`).
|
||||
//! * `Prefix`: review B's F05 text as landed (`--reg64-prefix`): a running `S` updated after every write (main and
|
||||
//! shadow), `P_s` recomputed per load.
|
||||
//! * `Wrong`: the known-failed variant for the differential test only (`r[s] ^ S ^ a[s]`, every term rotated as if it
|
||||
//! sat after the source); the differential must refuse it. Never emitted.
|
||||
|
||||
use std::sync::atomic::{AtomicU8, Ordering};
|
||||
|
||||
use crate::generator::{Instr, Op, LANES};
|
||||
|
||||
/// Registers per group of the generated form.
|
||||
pub const GROUP: usize = 8;
|
||||
/// Groups of the generated form (64 registers).
|
||||
pub const GROUPS: usize = 8;
|
||||
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum Reg64Form {
|
||||
Fold,
|
||||
Direct,
|
||||
Generated,
|
||||
Prefix,
|
||||
Wrong,
|
||||
}
|
||||
|
||||
impl Reg64Form {
|
||||
pub fn parse(s: &str) -> Option<Reg64Form> {
|
||||
Some(match s {
|
||||
"fold" => Reg64Form::Fold,
|
||||
"direct" => Reg64Form::Direct,
|
||||
"generated" => Reg64Form::Generated,
|
||||
"prefix" => Reg64Form::Prefix,
|
||||
"wrong" => Reg64Form::Wrong,
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
pub fn name(self) -> &'static str {
|
||||
match self {
|
||||
Reg64Form::Fold => "fold",
|
||||
Reg64Form::Direct => "direct",
|
||||
Reg64Form::Generated => "generated",
|
||||
Reg64Form::Prefix => "prefix",
|
||||
Reg64Form::Wrong => "wrong",
|
||||
}
|
||||
}
|
||||
fn code(self) -> u8 {
|
||||
self as u8
|
||||
}
|
||||
fn from_code(c: u8) -> Reg64Form {
|
||||
[Reg64Form::Fold, Reg64Form::Direct, Reg64Form::Generated, Reg64Form::Prefix, Reg64Form::Wrong][c as usize]
|
||||
}
|
||||
}
|
||||
|
||||
static GPU_FORM: AtomicU8 = AtomicU8::new(0);
|
||||
static CPU_FORM: AtomicU8 = AtomicU8::new(0);
|
||||
|
||||
/// The CUDA text's form (`igneum-pow export --reg64-form <form>`; `--reg64-prefix` is `prefix`). `Wrong` is refused.
|
||||
pub fn set_gpu_form(f: Reg64Form) {
|
||||
assert!(f != Reg64Form::Wrong, "the known-failed form is never emitted");
|
||||
GPU_FORM.store(f.code(), Ordering::Relaxed);
|
||||
}
|
||||
pub fn gpu_form() -> Reg64Form {
|
||||
Reg64Form::from_code(GPU_FORM.load(Ordering::Relaxed))
|
||||
}
|
||||
/// The CPU interpreter's form (`--reg64-cpu-form <form>`, the differential test's switch). The fold is the
|
||||
/// definition and the default; every other form runs beside it only in the X4 differential.
|
||||
pub fn set_cpu_form(f: Reg64Form) {
|
||||
CPU_FORM.store(f.code(), Ordering::Relaxed);
|
||||
}
|
||||
pub fn cpu_form() -> Reg64Form {
|
||||
Reg64Form::from_code(CPU_FORM.load(Ordering::Relaxed))
|
||||
}
|
||||
|
||||
/// `a[k] = rotl(r[k], (63 - k) mod 32)`: register k's rotation in the full 64-register chain.
|
||||
#[inline(always)]
|
||||
pub fn term(r: &[u32; 64], k: usize) -> u32 {
|
||||
r[k].rotate_left(((63 - k) % 32) as u32)
|
||||
}
|
||||
|
||||
/// The rotation register `k` carries in the chain of a load whose source is `s` (k != s).
|
||||
#[inline(always)]
|
||||
pub fn direct_rot(k: usize, s: usize) -> u32 {
|
||||
if k < s { ((62 - k) % 32) as u32 } else { ((63 - k) % 32) as u32 }
|
||||
}
|
||||
|
||||
/// The definition, one lane: `r[s] ^ m` with the 63-term chain (the same loop as `verify::step`).
|
||||
pub fn fold(r: &[u32; 64], s: usize) -> u32 {
|
||||
let mut m = 0u32;
|
||||
let mut started = false;
|
||||
for (k, &v) in r.iter().enumerate() {
|
||||
if k == s {
|
||||
continue;
|
||||
}
|
||||
m = if started { m.rotate_left(1) ^ v } else { v };
|
||||
started = true;
|
||||
}
|
||||
r[s] ^ m
|
||||
}
|
||||
|
||||
/// The direct form: the closed sum, one rotation per term.
|
||||
pub fn direct(r: &[u32; 64], s: usize) -> u32 {
|
||||
let mut m = 0u32;
|
||||
for k in 0..64 {
|
||||
if k != s {
|
||||
m ^= r[k].rotate_left(direct_rot(k, s));
|
||||
}
|
||||
}
|
||||
r[s] ^ m
|
||||
}
|
||||
|
||||
/// One group sum of the generated form.
|
||||
#[inline(always)]
|
||||
pub fn group(r: &[u32; 64], j: usize) -> u32 {
|
||||
let mut g = 0u32;
|
||||
for k in j * GROUP..(j + 1) * GROUP {
|
||||
g ^= term(r, k);
|
||||
}
|
||||
g
|
||||
}
|
||||
|
||||
/// The generated form's query from the cached group sums `g` (stale groups give a wrong word: the plan's job).
|
||||
pub fn grouped(g: &[u32; GROUPS], r: &[u32; 64], s: usize) -> u32 {
|
||||
let total = g.iter().fold(0u32, |x, y| x ^ y);
|
||||
let j = s / GROUP;
|
||||
let mut p = g[..j].iter().fold(0u32, |x, y| x ^ y);
|
||||
for k in j * GROUP..s {
|
||||
p ^= term(r, k);
|
||||
}
|
||||
r[s] ^ p.rotate_right(1) ^ total ^ p ^ term(r, s)
|
||||
}
|
||||
|
||||
/// The prefix form's query from a running total `total` (= XOR of every `a[k]` when kept fresh).
|
||||
pub fn prefixed(total: u32, r: &[u32; 64], s: usize) -> u32 {
|
||||
let mut p = 0u32;
|
||||
for k in 0..s {
|
||||
p ^= term(r, k);
|
||||
}
|
||||
r[s] ^ p.rotate_right(1) ^ total ^ p ^ term(r, s)
|
||||
}
|
||||
|
||||
/// The known-failed form (the F05 test's wrong variant).
|
||||
pub fn wrong(r: &[u32; 64], s: usize) -> u32 {
|
||||
let total = (0..64).map(|k| term(r, k)).fold(0, |x, y| x ^ y);
|
||||
r[s] ^ total ^ term(r, s)
|
||||
}
|
||||
|
||||
/// The generated form's static plan: for each statement of the scheduled list, the groups to recompute just before
|
||||
/// it (non-empty only at loads). The groups entering an iteration are those written after the last load of the
|
||||
/// previous iteration's list plus every group the shadow block writes (group 0 on the shipped class); the kernel
|
||||
/// computes all eight after the register init, so iteration 0 recomputes a superset (idempotent).
|
||||
pub fn generated_plan(scheduled: &[Instr], shadow: &[Instr]) -> Vec<Vec<usize>> {
|
||||
let all: u8 = 0xff;
|
||||
let bit = |ins: &Instr| 1u8 << (ins.dst as usize / GROUP);
|
||||
let mut dirty = all;
|
||||
for ins in scheduled {
|
||||
if ins.op == Op::Load {
|
||||
dirty = 0;
|
||||
}
|
||||
dirty |= bit(ins);
|
||||
}
|
||||
let shadow_groups = shadow.iter().fold(0u8, |m, ins| m | bit(ins));
|
||||
let mut dirty = dirty | shadow_groups;
|
||||
let mut plan = Vec::with_capacity(scheduled.len());
|
||||
for ins in scheduled {
|
||||
if ins.op == Op::Load {
|
||||
plan.push((0..GROUPS).filter(|j| dirty & (1 << j) != 0).collect());
|
||||
dirty = 0;
|
||||
} else {
|
||||
plan.push(Vec::new());
|
||||
}
|
||||
dirty |= bit(ins);
|
||||
}
|
||||
plan
|
||||
}
|
||||
|
||||
/// The groups the shadow block writes (a bit per group).
|
||||
pub fn shadow_group_mask(shadow: &[Instr]) -> u8 {
|
||||
shadow.iter().fold(0u8, |m, ins| m | (1u8 << (ins.dst as usize / GROUP)))
|
||||
}
|
||||
|
||||
/// Counts of the generated plan per iteration: (loads, group recomputes) — the op count's input.
|
||||
pub fn plan_counts(plan: &[Vec<usize>], scheduled: &[Instr]) -> (usize, usize) {
|
||||
let loads = scheduled.iter().filter(|i| i.op == Op::Load).count();
|
||||
let recomputes = plan.iter().map(|v| v.len()).sum();
|
||||
(loads, recomputes)
|
||||
}
|
||||
|
||||
/// The CPU interpreter's state for a form (the differential's emulation of each kernel text): the generated
|
||||
/// form's eight cached group sums per lane, recomputed only where [`generated_plan`] says; the prefix form's running
|
||||
/// total per lane, updated after every write (main list and shadow), as the F05 text does.
|
||||
pub struct FormState {
|
||||
form: Reg64Form,
|
||||
plan: Vec<Vec<usize>>,
|
||||
groups: [[u32; LANES]; GROUPS],
|
||||
total: [u32; LANES],
|
||||
}
|
||||
|
||||
fn lane_regs(r: &[[u32; LANES]], lane: usize) -> [u32; 64] {
|
||||
std::array::from_fn(|k| r[k][lane])
|
||||
}
|
||||
|
||||
impl FormState {
|
||||
pub fn new(form: Reg64Form, plan: Vec<Vec<usize>>) -> FormState {
|
||||
FormState { form, plan, groups: [[0; LANES]; GROUPS], total: [0; LANES] }
|
||||
}
|
||||
/// After the register init (and the probe's flip): every group and the running total from the live registers.
|
||||
pub fn init(&mut self, r: &[[u32; LANES]]) {
|
||||
for lane in 0..LANES {
|
||||
let regs = lane_regs(r, lane);
|
||||
for j in 0..GROUPS {
|
||||
self.groups[j][lane] = group(®s, j);
|
||||
}
|
||||
self.total[lane] = (0..64).map(|k| term(®s, k)).fold(0, |x, y| x ^ y);
|
||||
}
|
||||
}
|
||||
/// After a write of register `d` whose value before the write was `old`: the prefix form's running total moves
|
||||
/// by the term's change; the generated form keeps its groups stale until the plan recomputes them.
|
||||
pub fn after_write(&mut self, r: &[[u32; LANES]], d: usize, old: &[u32; LANES]) {
|
||||
if self.form == Reg64Form::Prefix {
|
||||
let rot = ((63 - d) % 32) as u32;
|
||||
for lane in 0..LANES {
|
||||
self.total[lane] ^= old[lane].rotate_left(rot) ^ r[d][lane].rotate_left(rot);
|
||||
}
|
||||
}
|
||||
}
|
||||
/// The address source of the load at scheduled position `k` with source `s`, per lane, in this form.
|
||||
pub fn source(&mut self, r: &[[u32; LANES]], k: usize, s: usize) -> [u32; LANES] {
|
||||
if self.form == Reg64Form::Generated {
|
||||
for &j in &self.plan[k] {
|
||||
for lane in 0..LANES {
|
||||
let regs = lane_regs(r, lane);
|
||||
self.groups[j][lane] = group(®s, j);
|
||||
}
|
||||
}
|
||||
}
|
||||
std::array::from_fn(|lane| {
|
||||
let regs = lane_regs(r, lane);
|
||||
match self.form {
|
||||
Reg64Form::Fold => fold(®s, s),
|
||||
Reg64Form::Direct => direct(®s, s),
|
||||
Reg64Form::Generated => grouped(&std::array::from_fn(|j| self.groups[j][lane]), ®s, s),
|
||||
Reg64Form::Prefix => prefixed(self.total[lane], ®s, s),
|
||||
Reg64Form::Wrong => wrong(®s, s),
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn stream(seed: u64) -> impl FnMut() -> u32 {
|
||||
let mut x = seed;
|
||||
move || {
|
||||
x = x.wrapping_add(0x9e37_79b9_7f4a_7c15);
|
||||
let mut z = x;
|
||||
z = (z ^ (z >> 30)).wrapping_mul(0xbf58_476d_1ce4_e5b9);
|
||||
z = (z ^ (z >> 27)).wrapping_mul(0x94d0_49bb_1331_11eb);
|
||||
(z ^ (z >> 31)) as u32
|
||||
}
|
||||
}
|
||||
|
||||
/// Known-failed first: the wrong form must disagree with the fold somewhere; then the three forms agree with the
|
||||
/// fold on every source over 4,096 random states and on the structured states (all zero, all ones, one bit,
|
||||
/// the init rule's registers).
|
||||
#[test]
|
||||
fn x4_forms_equal_the_fold_and_the_wrong_form_is_refused() {
|
||||
let mut next = stream(0x1505_0004);
|
||||
let mut wrong_differs = 0usize;
|
||||
let mut states: Vec<[u32; 64]> = Vec::new();
|
||||
for _ in 0..4096 {
|
||||
let mut r = [0u32; 64];
|
||||
for v in r.iter_mut() {
|
||||
*v = next();
|
||||
}
|
||||
states.push(r);
|
||||
}
|
||||
states.push([0u32; 64]);
|
||||
states.push([u32::MAX; 64]);
|
||||
for b in 0..64 * 32 {
|
||||
let mut r = [0u32; 64];
|
||||
r[b / 32] = 1 << (b % 32);
|
||||
states.push(r);
|
||||
}
|
||||
let mut r = [0u32; 64];
|
||||
for k in 0..8 {
|
||||
r[k] = next();
|
||||
}
|
||||
for k in 8..64 {
|
||||
r[k] = r[k & 7].wrapping_mul(0x9e37_79b9).wrapping_add(k as u32);
|
||||
}
|
||||
states.push(r);
|
||||
for r in &states {
|
||||
let g: [u32; GROUPS] = std::array::from_fn(|j| group(r, j));
|
||||
let total = (0..64).map(|k| term(r, k)).fold(0, |x, y| x ^ y);
|
||||
for s in 0..64 {
|
||||
let want = fold(r, s);
|
||||
if wrong(r, s) != want {
|
||||
wrong_differs += 1;
|
||||
}
|
||||
assert_eq!(direct(r, s), want, "direct, source r{s}");
|
||||
assert_eq!(grouped(&g, r, s), want, "generated, source r{s}");
|
||||
assert_eq!(prefixed(total, r, s), want, "prefix, source r{s}");
|
||||
}
|
||||
}
|
||||
assert!(wrong_differs > 0, "the known-failed form must be refused");
|
||||
}
|
||||
}
|
||||
|
|
@ -581,6 +581,10 @@ fn interpret_warp_core(
|
|||
// the iteration's statements: the drawn program, or its interleaved two-window form under reg64
|
||||
let scheduled = program.scheduled();
|
||||
let address_mix = program.address_mix();
|
||||
// X4: a byte-identical form of the address mix beside the definition (crate::reg64form), off by default
|
||||
let cpu_form = crate::reg64form::cpu_form();
|
||||
let form_on = address_mix && cpu_form != crate::reg64form::Reg64Form::Fold;
|
||||
let mut fs = crate::reg64form::FormState::new(cpu_form, if form_on { crate::reg64form::generated_plan(&scheduled, &program.shadow) } else { Vec::new() });
|
||||
let mut items_derived = 0usize;
|
||||
let mut idx = [0u32; LANES];
|
||||
let mut val = [0u32; LANES];
|
||||
|
|
@ -606,9 +610,15 @@ fn interpret_warp_core(
|
|||
}
|
||||
}
|
||||
}
|
||||
if it == 0 && form_on {
|
||||
// X4: the form's state from the registers as iteration 0 sees them (after the init and any probe flip)
|
||||
fs.init(&r);
|
||||
}
|
||||
let sel = r[0];
|
||||
for ins in &scheduled {
|
||||
step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix);
|
||||
for (k, ins) in scheduled.iter().enumerate() {
|
||||
let old = if form_on { r[ins.dst as usize] } else { [0u32; LANES] };
|
||||
let ovr = if form_on && ins.op == Op::Load { Some(fs.source(&r, k, ins.src as usize)) } else { None };
|
||||
step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix, ovr.as_ref());
|
||||
if it == 0 && ins.op == Op::Load {
|
||||
if let Some(pr) = probe.as_deref_mut() {
|
||||
if !pr.seen_first_load {
|
||||
|
|
@ -625,12 +635,19 @@ fn interpret_warp_core(
|
|||
r[d][lane] = m.rmw(&program.seed, base_nonce, lane, slot, r[d][lane]);
|
||||
}
|
||||
}
|
||||
if form_on {
|
||||
fs.after_write(&r, ins.dst as usize, &old);
|
||||
}
|
||||
}
|
||||
// Latency-shadow block (Counter ASIC 3.0 item 8): the block runs `reps` times after instruction 63 with the
|
||||
// iteration's `sel`; it is empty on every class without a shadow, so version 2 and class v3 run nothing here.
|
||||
for _ in 0..program.shadow_reps() {
|
||||
for ins in &program.shadow {
|
||||
step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix);
|
||||
let old = if form_on { r[ins.dst as usize] } else { [0u32; LANES] };
|
||||
step(ins, &mut r, &sel, geom, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived, address_mix, None);
|
||||
if form_on {
|
||||
fs.after_write(&r, ins.dst as usize, &old);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -668,6 +685,7 @@ fn step(
|
|||
val: &mut [u32; LANES],
|
||||
items_derived: &mut usize,
|
||||
address_mix: bool,
|
||||
src_override: Option<&[u32; LANES]>,
|
||||
) {
|
||||
let d = ins.dst as usize;
|
||||
let a = ins.src as usize;
|
||||
|
|
@ -679,6 +697,11 @@ fn step(
|
|||
if !address_mix {
|
||||
return r[a][lane];
|
||||
}
|
||||
// X4 (1.5x programme): a byte-identical form computed by the caller; never set unless the differential's
|
||||
// `--reg64-cpu-form` names a form other than the fold (crate::reg64form)
|
||||
if let Some(o) = src_override {
|
||||
return o[lane];
|
||||
}
|
||||
let mut m = 0u32;
|
||||
let mut started = false;
|
||||
for k in 0..r.len() {
|
||||
|
|
|
|||
Loading…
Reference in a new issue