igneum/tools/attack/f8-uniform/src/main.rs
igneum-labs 6033df803c adv-mixer: attack plan for the memory-hard mixer M_r, harnesses copied
Internal adversarial pass, not an independent review. The plan restates M_r
from the frozen crate and spec 1.8, ranks Q3/Q2/Q1/Q4, gives the method, the
planted known-fail shape and the box-hour estimate per step, and lists every
file opened. Records the byte-identity result: build/master differs from the
frozen commit in accept.rs, emit.rs, generator.rs, packcheck.rs and two test
files, but is byte-identical over memhard.rs, seed.rs, bind.rs, derive.rs and
the Cargo pin, which define M_r.

Harnesses: adv-mixer (new: diffusion margin for Q2, the fold probe for Q1,
with a planted-weak-day hook), f4-weakday (copied read-only from
build/attack-pass for the Q3 census), f8-uniform (copied from the regate
worktree).

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-07 18:09:36 +00:00

1745 lines
73 KiB
Rust

//! attack-f8: the uniformity censuses of attack-pass row F8 (`docs/plans/cryptanalysis.md` 4.2 F8) on the real
//! class v4 derivation through the `igneum-pow` library.
//!
//! Three censuses, one binary:
//!
//! * `lines`: the cache-line index `s[0] AND line_mask` (MEMHARD.md 1.6, 2^22 lines) of every one of the 8 reads of
//! every item `t < 2^items_log2` on `days` consecutive day keys: the full 2^22-line histogram per round and
//! pooled, the 2^16-bucket (64-line segment) histogram, the largest bucket in sigma, chi-square, and a uniform
//! SplitMix64 control of the same size. Every item is also derived by the library's `derive_items` and compared
//! word for word (the traced mirror is trusted only while that holds).
//! * `warps`: the warp interpreter mirrored with every load's item index and the item's 8 lines recorded (the
//! day's items derived once into a table, as a GPU holds the dataset): distinct lines and items per hash and per
//! warp over `nonces` nonces of one program, the cross-hash item histogram with the hot-set test, and a uniform
//! control. Sampled warps are hashed by `Epoch::hash_warp` as well and must agree bit for bit.
//!
//! Plant hooks (the known-fail firings): `--plant quarter-lines` masks the line index to a quarter of its range,
//! `--plant half-lines` to a half, `--plant const-item` feeds one constant item at the program's first load site.
//! Under a plant the library comparisons are skipped (the plant is not the derivation) and the log says so.
//!
//! Nothing in `igneum-pow` is modified; every constant and the read sequence are the library's.
use igneum_pow::bind::{day_bytes, hex, unhex};
use igneum_pow::generator::{EraParams, Instr, Op, Program, ProgramClass, ITERATIONS, LANES};
use igneum_pow::memhard::{mixer, round_key_mult, Cache, Layout, MixParams, ITEM_ROUNDS};
use igneum_pow::seed::{seed_words_from_bytes, SplitMix64};
use igneum_pow::verify::{load_index, splitmix32, DatasetSource, Epoch};
use std::io::Write;
use std::sync::atomic::{AtomicU32, AtomicUsize, Ordering};
use std::time::{Instant, SystemTime, UNIX_EPOCH};
/// The devnet genesis hash: epoch seed and era seed of the chain's epoch 0 (`igneum-pow/tests/packs.rs`).
const GENESIS_HEX: &str = "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07";
/// The devnet pack's day index (`bind::day_bytes(20730)`, 2026-10-04).
const DEFAULT_DAY: u64 = 20730;
const DATASET_LOG2: u32 = 28;
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
enum Plant {
None,
QuarterLines,
HalfLines,
ConstItem,
}
impl Plant {
fn parse(s: &str) -> Plant {
match s {
"none" => Plant::None,
"quarter-lines" => Plant::QuarterLines,
"half-lines" => Plant::HalfLines,
"const-item" => Plant::ConstItem,
_ => panic!("unknown plant {s}"),
}
}
fn line_mask(self) -> u32 {
match self {
Plant::QuarterLines => (1u32 << 20) - 1,
Plant::HalfLines => (1u32 << 21) - 1,
_ => u32::MAX,
}
}
fn name(self) -> &'static str {
match self {
Plant::None => "none",
Plant::QuarterLines => "quarter-lines",
Plant::HalfLines => "half-lines",
Plant::ConstItem => "const-item",
}
}
}
fn utc_now() -> String {
let s = SystemTime::now().duration_since(UNIX_EPOCH).unwrap().as_secs();
let (d, t) = (s / 86400, s % 86400);
// civil date from days (Howard Hinnant's algorithm)
let z = d as i64 + 719468;
let era = z.div_euclid(146097);
let doe = z.rem_euclid(146097);
let yoe = (doe - doe / 1460 + doe / 36524 - doe / 146096) / 365;
let y = yoe + era * 400;
let doy = doe - (365 * yoe + yoe / 4 - yoe / 100);
let mp = (5 * doy + 2) / 153;
let dd = doy - (153 * mp + 2) / 5 + 1;
let mm = if mp < 10 { mp + 3 } else { mp - 9 };
let yy = if mm <= 2 { y + 1 } else { y };
format!("{yy:04}-{mm:02}-{dd:02}T{:02}:{:02}:{:02}Z", t / 3600, (t / 60) % 60, t % 60)
}
macro_rules! log {
($($arg:tt)*) => {{
println!("[{}] {}", utc_now(), format!($($arg)*));
std::io::stdout().flush().ok();
}};
}
// --------------------------------------------------------------------------------------------------------------
// The traced item derivation: `memhard::derive_items_mask` instruction for instruction, with the line index of
// every round recorded. Batched like the library so the 8 dependent misses of independent items overlap.
// --------------------------------------------------------------------------------------------------------------
fn derive_traced(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]], lines: &mut [[u32; 8]], plant_mask: u32) {
let n = ts.len();
let m = mp.shape.mixer_mult as usize;
assert_eq!(mp.shape.derive_len, 0, "class v4 has the fixed mixer");
let line_mask = cache.line_mask();
for k in 0..n {
let s = &mut out[k];
let t = ts[k];
s[..8].copy_from_slice(&mp.key);
for i in 0..8 {
s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]);
}
}
for r in 0..ITEM_ROUNDS {
for j in 0..m {
let rk = round_key_mult(r, j, m);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
}
}
for k in 0..n {
let s = &mut out[k];
// MEMHARD.md 1.6: `a = s[0] AND 0x003fffff`, the cache line index (the mask is the cache's at every size)
let a = s[0] & line_mask & plant_mask;
lines[k][r] = a;
let line = cache.line(a);
for i in 0..16 {
s[i] ^= line[i];
}
}
}
for j in 0..m {
let rk = round_key_mult(ITEM_ROUNDS, j, m);
for s in out[..n].iter_mut() {
mixer(s, rk, mp);
}
}
}
// --------------------------------------------------------------------------------------------------------------
// Histogram statistics
// --------------------------------------------------------------------------------------------------------------
struct Stats {
bins: usize,
total: u64,
mean: f64,
sigma: f64,
max: u64,
argmax: usize,
min: u64,
z_max: f64,
z_min: f64,
chi2_per_dof: f64,
chi2_z: f64,
/// (fraction f, top-f share S_f) for f = 1/1000, 1/200, 1/100
top: [(f64, f64); 3],
}
fn stats_of(counts: impl Iterator<Item = u64> + Clone, bins: usize) -> Stats {
let total: u64 = counts.clone().sum();
let mean = total as f64 / bins as f64;
let sigma = mean.sqrt();
let mut max = 0u64;
let mut argmax = 0usize;
let mut min = u64::MAX;
let mut chi2 = 0f64;
for (i, c) in counts.clone().enumerate() {
if c > max {
max = c;
argmax = i;
}
if c < min {
min = c;
}
let d = c as f64 - mean;
chi2 += d * d;
}
chi2 /= mean;
let dof = (bins - 1) as f64;
// count of counts for the top-f shares (counts at or above the cap, rare, are sorted on their own)
let cap = 1usize << 16;
let mut coc = vec![0u64; cap];
let mut big: Vec<u64> = Vec::new();
for c in counts {
if (c as usize) < cap {
coc[c as usize] += 1;
} else {
big.push(c);
}
}
big.sort_unstable_by(|a, b| b.cmp(a));
let mut top = [(0f64, 0f64); 3];
for (k, f) in [1.0 / 1000.0, 1.0 / 200.0, 1.0 / 100.0].into_iter().enumerate() {
let want = (f * bins as f64).round() as u64;
let mut left = want;
let mut reads = 0u64;
for &c in &big {
if left == 0 {
break;
}
reads += c;
left -= 1;
}
let mut v = cap - 1;
while left > 0 {
let n = coc[v].min(left);
reads += n * v as u64;
left -= n;
if v == 0 {
break;
}
v -= 1;
}
top[k] = (f, reads as f64 / total.max(1) as f64);
}
Stats {
bins,
total,
mean,
sigma,
max,
argmax,
min,
z_max: (max as f64 - mean) / sigma,
z_min: (min as f64 - mean) / sigma,
chi2_per_dof: chi2 / dof,
chi2_z: (chi2 - dof) / (2.0 * dof).sqrt(),
top,
}
}
impl Stats {
fn line(&self, label: &str) -> String {
format!(
"{label}: bins {} reads {} mean {:.3} sigma {:.3} max {} (bin {}) z_max {:+.2} min {} z_min {:+.2} chi2/dof {:.5} chi2_z {:+.2} top0.1% {:.5}% top0.5% {:.5}% top1% {:.5}%",
self.bins,
self.total,
self.mean,
self.sigma,
self.max,
self.argmax,
self.z_max,
self.min,
self.z_min,
self.chi2_per_dof,
self.chi2_z,
self.top[0].1 * 100.0,
self.top[1].1 * 100.0,
self.top[2].1 * 100.0
)
}
}
fn snapshot(counts: &[AtomicU32]) -> Vec<u64> {
counts.iter().map(|x| x.load(Ordering::Relaxed) as u64).collect()
}
fn zero(counts: &[AtomicU32]) {
for c in counts {
c.store(0, Ordering::Relaxed);
}
}
fn atomic_vec(n: usize) -> Vec<AtomicU32> {
(0..n).map(|_| AtomicU32::new(0)).collect()
}
/// A uniform control: `n` SplitMix64 indices into `bins` bins, `threads` threads.
fn control(bins: usize, n: u64, seed: u64, threads: usize, hist: &[AtomicU32]) {
let mask = (bins - 1) as u64;
assert!(bins.is_power_of_two());
std::thread::scope(|sc| {
for th in 0..threads {
let hist = &hist;
sc.spawn(move || {
let mut s = SplitMix64::new(seed ^ (th as u64).wrapping_mul(0x9E3779B97F4A7C15));
let per = n / threads as u64 + if (th as u64) < n % threads as u64 { 1 } else { 0 };
for _ in 0..per {
let i = (s.next() & mask) as usize;
hist[i].fetch_add(1, Ordering::Relaxed);
}
});
}
});
}
/// The hot-set test (F8's definition, reused by F9): for the top-f items by count, the share S_f of all reads
/// they receive, against the same share E_f of a uniform control of the same size; the excess X_f = S_f - E_f.
/// The set is a hot set when X_f >= f for any f in {0.1%, 0.5%, 1%}: after the chance excess is removed, the top
/// f of items capture at least one extra proportional share (at least 2f of the reads above chance, which is
/// what a chip's on-die copy of f of the items would have to win to matter). The statistical sensitivity is
/// printed beside it: X_f in units of f.
fn hot_set_test(label: &str, real: &Stats, ctrl: &Stats) -> bool {
let mut flagged = false;
for k in 0..3 {
let (f, s) = real.top[k];
let e = ctrl.top[k].1;
let x = s - e;
let fire = x >= f;
flagged |= fire;
log!(
"hot-set {label}: f {:.1}% S_f {:.5}% E_f(control) {:.5}% X_f {:+.5}% X_f/f {:+.4} -> {}",
f * 100.0,
s * 100.0,
e * 100.0,
x * 100.0,
x / f,
if fire { "HOT SET" } else { "no hot set" }
);
}
log!("hot-set {label}: verdict {}", if flagged { "FLAGGED" } else { "clear" });
flagged
}
/// The 6-sigma test: the largest bucket within 6 sigma of uniform.
fn six_sigma_test(label: &str, s: &Stats) -> bool {
let fire = s.z_max > 6.0 || s.z_min < -6.0;
log!(
"6-sigma {label}: largest bucket {} at {:+.2} sigma, smallest {} at {:+.2} sigma -> {}",
s.max,
s.z_max,
s.min,
s.z_min,
if fire { "FLAGGED (beyond 6 sigma)" } else { "within 6 sigma" }
);
fire
}
// --------------------------------------------------------------------------------------------------------------
// Census 1: the line index over all items of `days` day keys
// --------------------------------------------------------------------------------------------------------------
fn census_lines(day0: u64, days: u64, items_log2: u32, threads: usize, plant: Plant, validate: &str, out_dir: &str) {
log!("census lines: day0 {day0} days {days} items 2^{items_log2} per day threads {threads} plant {} validate {validate}", plant.name());
let lines_n = 1usize << 22;
let per_round: Vec<Vec<AtomicU32>> = (0..ITEM_ROUNDS).map(|_| atomic_vec(lines_n)).collect();
let pooled: Vec<Vec<AtomicU32>> = (0..ITEM_ROUNDS).map(|_| atomic_vec(lines_n)).collect();
let items = 1u64 << items_log2;
let batch = 64u64;
let mut any_fire = false;
let mut mismatches_total = 0u64;
let mut derivations = 0u64;
let t_all = Instant::now();
for d in day0..day0 + days {
let t0 = Instant::now();
let ds = Epoch::chain_dataset_day(&day_bytes(d), ProgramClass::V4, 0, DATASET_LOG2);
let mh = ds.memhard().expect("memory-hard");
assert_eq!(mh.params.shape.mixer_mult, 8);
assert_eq!(mh.params.shape.cache_log2_words, 26);
assert_eq!(mh.cache.line_mask(), (1u32 << 22) - 1);
log!(
"day {d}: day bytes {} key {} rot {:?} cache 2^26 words fnv {:016x} filled in {:.2} s",
hex(&day_bytes(d)),
mh.params.key.iter().map(|w| format!("{w:08x}")).collect::<Vec<_>>().join(""),
mh.params.rot,
mh.cache.fnv1a64(),
t0.elapsed().as_secs_f64()
);
for h in &per_round {
zero(h);
}
let next = AtomicUsize::new(0);
let mism = AtomicUsize::new(0);
let t1 = Instant::now();
std::thread::scope(|sc| {
for _ in 0..threads {
let (next, mism, per_round, mp, cache) = (&next, &mism, &per_round, &mh.params, &mh.cache);
sc.spawn(move || {
let mut ts = [0u32; 64];
let mut out = [[0u32; 16]; 64];
let mut lines = [[0u32; 8]; 64];
let mut lib = [[0u32; 16]; 64];
loop {
let b = next.fetch_add(1, Ordering::Relaxed) as u64;
let start = b * batch;
if start >= items {
break;
}
for k in 0..64 {
ts[k] = (start + k as u64) as u32;
}
derive_traced(&ts, mp, cache, &mut out, &mut lines, plant.line_mask());
for k in 0..64 {
for r in 0..ITEM_ROUNDS {
per_round[r][lines[k][r] as usize].fetch_add(1, Ordering::Relaxed);
}
}
let check = plant == Plant::None && (validate == "all" || (validate == "sample" && b % 64 == 0));
if check {
igneum_pow::memhard::derive_items(&ts, mp, cache, &mut lib);
for k in 0..64 {
if lib[k] != out[k] {
mism.fetch_add(1, Ordering::Relaxed);
}
}
}
}
});
}
});
let el = t1.elapsed().as_secs_f64();
derivations += items;
let mm = mism.load(Ordering::Relaxed) as u64;
mismatches_total += mm;
log!(
"day {d}: {items} items ({} line reads) derived in {el:.1} s ({:.2} M items/s); library comparison: {} ({} mismatches)",
items * 8,
items as f64 / el / 1e6,
if plant == Plant::None { validate } else { "skipped under the plant" },
mm
);
if mm > 0 {
log!("day {d}: THE TRACED MIRROR DISAGREES WITH THE LIBRARY: {mm} items; nothing below is trusted");
}
// per-day stats: per round, total over rounds, buckets of 64 lines (one segment)
let mut total = vec![0u64; lines_n];
for r in 0..ITEM_ROUNDS {
let snap = snapshot(&per_round[r]);
let s = stats_of(snap.iter().copied(), lines_n);
log!("{}", s.line(&format!("day {d} round {r} lines")));
for (i, c) in snap.iter().enumerate() {
total[i] += c;
pooled[r][i].fetch_add(*c as u32, Ordering::Relaxed);
}
}
let s_full = stats_of(total.iter().copied(), lines_n);
log!("{}", s_full.line(&format!("day {d} all rounds lines")));
let b: Vec<u64> = total.chunks(64).map(|c| c.iter().sum()).collect();
let s_b = stats_of(b.iter().copied(), b.len());
log!("{}", s_b.line(&format!("day {d} all rounds buckets64")));
let fire = six_sigma_test(&format!("day {d} buckets64"), &s_b);
any_fire |= fire;
// control of the same size
let ctrl = atomic_vec(lines_n);
control(lines_n, items * 8, 0xF8_0000 + d, threads, &ctrl);
let cs = snapshot(&ctrl);
let c_full = stats_of(cs.iter().copied(), lines_n);
log!("{}", c_full.line(&format!("day {d} CONTROL lines")));
let cb: Vec<u64> = cs.chunks(64).map(|c| c.iter().sum()).collect();
let c_b = stats_of(cb.iter().copied(), cb.len());
log!("{}", c_b.line(&format!("day {d} CONTROL buckets64")));
let hot = hot_set_test(&format!("day {d} lines"), &s_full, &c_full);
any_fire |= hot;
}
// pooled over the days
let mut total = vec![0u64; lines_n];
for r in 0..ITEM_ROUNDS {
let snap = snapshot(&pooled[r]);
let s = stats_of(snap.iter().copied(), lines_n);
log!("{}", s.line(&format!("POOLED {days} days round {r} lines")));
for (i, c) in snap.iter().enumerate() {
total[i] += c;
}
}
let s_full = stats_of(total.iter().copied(), lines_n);
log!("{}", s_full.line(&format!("POOLED {days} days all rounds lines")));
let b: Vec<u64> = total.chunks(64).map(|c| c.iter().sum()).collect();
let s_b = stats_of(b.iter().copied(), b.len());
log!("{}", s_b.line(&format!("POOLED {days} days all rounds buckets64")));
let fire = six_sigma_test(&format!("POOLED {days} days buckets64"), &s_b);
let fire_full = six_sigma_test(&format!("POOLED {days} days full 2^22 lines"), &s_full);
let ctrl = atomic_vec(lines_n);
control(lines_n, derivations * 8, 0xF8_1111, threads, &ctrl);
let cs = snapshot(&ctrl);
let c_full = stats_of(cs.iter().copied(), lines_n);
log!("{}", c_full.line(&format!("POOLED CONTROL lines")));
let cb: Vec<u64> = cs.chunks(64).map(|c| c.iter().sum()).collect();
let c_b = stats_of(cb.iter().copied(), cb.len());
log!("{}", c_b.line(&format!("POOLED CONTROL buckets64")));
let hot = hot_set_test(&format!("POOLED {days} days lines"), &s_full, &c_full);
any_fire |= fire | fire_full | hot;
// files
let tag = format!("lines-d{day0}-n{days}-i{items_log2}-{}", plant.name());
{
let mut f = std::fs::File::create(format!("{out_dir}/{tag}-buckets64.txt")).unwrap();
writeln!(f, "# bucket(64 lines = one segment) count ; pooled over {days} days from {day0}, items 2^{items_log2} per day, plant {}", plant.name()).unwrap();
for (i, c) in b.iter().enumerate() {
writeln!(f, "{i} {c}").unwrap();
}
let mut g = std::fs::File::create(format!("{out_dir}/{tag}-full.u32le")).unwrap();
let mut bytes = Vec::with_capacity(lines_n * 4);
for c in &total {
bytes.extend_from_slice(&(*c as u32).to_le_bytes());
}
g.write_all(&bytes).unwrap();
}
log!(
"census lines DONE: {derivations} item derivations ({} line reads) over {days} days in {:.1} s; mirror mismatches {mismatches_total}; verdict {}",
derivations * 8,
t_all.elapsed().as_secs_f64(),
if any_fire { "FLAGGED" } else { "PASS (no test fired)" }
);
}
// --------------------------------------------------------------------------------------------------------------
// Census 2 and 3: the warp interpreter mirrored with every item index recorded
// --------------------------------------------------------------------------------------------------------------
struct Table {
items: Vec<[u32; 16]>,
lines: Vec<[u32; 8]>,
}
fn build_table(mp: &MixParams, cache: &Cache, threads: usize, plant: Plant, validate: &str) -> (Table, u64) {
let n = 1usize << (DATASET_LOG2 - 4);
let t0 = Instant::now();
let mut items = vec![[0u32; 16]; n];
let mut lines = vec![[0u32; 8]; n];
let mism = AtomicUsize::new(0);
std::thread::scope(|sc| {
let per = n / threads;
for (th, (ic, lc)) in items.chunks_mut(per).zip(lines.chunks_mut(per)).enumerate() {
let mism = &mism;
sc.spawn(move || {
let mut ts = [0u32; 64];
let mut lib = [[0u32; 16]; 64];
let base = th * per;
for (bi, (ib, lb)) in ic.chunks_mut(64).zip(lc.chunks_mut(64)).enumerate() {
let start = base + bi * 64;
for k in 0..ib.len() {
ts[k] = (start + k) as u32;
}
derive_traced(&ts[..ib.len()], mp, cache, ib, lb, plant.line_mask());
let check = plant == Plant::None && (validate == "all" || (validate == "sample" && bi % 64 == 0));
if check {
igneum_pow::memhard::derive_items(&ts[..ib.len()], mp, cache, &mut lib);
for k in 0..ib.len() {
if lib[k] != ib[k] {
mism.fetch_add(1, Ordering::Relaxed);
}
}
}
}
});
}
});
let mm = mism.load(Ordering::Relaxed) as u64;
log!(
"table: {n} items derived with their 8 lines in {:.1} s; library comparison {} ({mm} mismatches)",
t0.elapsed().as_secs_f64(),
if plant == Plant::None { validate } else { "skipped under the plant" }
);
(Table { items, lines }, mm)
}
struct ProgramSpec {
label: String,
epoch_seed: Vec<u8>,
era: Vec<u8>,
}
fn program_spec(k: u32) -> ProgramSpec {
let genesis = unhex(GENESIS_HEX).unwrap();
match k {
1 => ProgramSpec { label: "p1-devnet-epoch0".into(), epoch_seed: genesis.clone(), era: genesis },
_ => {
let w = |s: String| -> Vec<u8> { seed_words_from_bytes(s.as_bytes()).iter().flat_map(|x| x.to_le_bytes()).collect() };
ProgramSpec {
label: format!("p{k}-attack-f8"),
epoch_seed: w(format!("igneum-attack-f8/program/{k}")),
era: w(format!("igneum-attack-f8/era/{k}")),
}
}
}
}
/// The mirror of `verify::interpret_warp_init` for class v4 (the ops a v4 program can hold), reading dataset words
/// from the item table and reporting every load's item index.
/// The mirror of `verify::interpret_warp_init` for class v4 (the ops a v4 program can hold), reading dataset words
/// from the item table and reporting every load's item index and, per load position, how many lanes' source
/// register was saturated (0 or all ones) at the load.
struct Mirror<'a> {
program: &'a Program,
era: Option<&'a EraParams>,
layout: Layout,
mask: u32,
log2: u32,
table: &'a Table,
plant_site: Option<usize>,
}
const POS: usize = 128;
struct Sink {
/// the item index of every load position, per lane
items: Vec<[u32; LANES]>,
/// the masked word index (the address before the layout split) of every load position, per lane
addrs: Vec<[u32; LANES]>,
/// lanes whose source register read 0 or 2^32 - 1 at the load, per position
sat: [u32; POS],
}
impl Sink {
fn new() -> Sink {
Sink { items: Vec::with_capacity(POS), addrs: Vec::with_capacity(POS), sat: [0; POS] }
}
}
impl<'a> Mirror<'a> {
#[inline(always)]
fn step(&self, ins: &Instr, r: &mut [[u32; LANES]; 8], sel: &[u32; LANES], sink: &mut Sink, site: &mut usize) {
let d = ins.dst as usize;
let a = ins.src as usize;
match ins.op {
Op::Add => {
let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32);
let src = r[a];
for lane in 0..LANES {
let s = (sel[lane] >> bit) & 1;
let c = if s != 0 { imm2 } else { imm };
r[d][lane] = r[d][lane].wrapping_add(src[lane]).wrapping_add(c);
}
}
Op::Sub => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_sub(src[lane]);
}
}
Op::Mul => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_mul(src[lane]);
}
}
Op::MulHi => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = ((r[d][lane] as u64 * src[lane] as u64) >> 32) as u32;
}
}
Op::Xor => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] ^= src[lane];
}
}
Op::Or => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] |= src[lane];
}
}
Op::Rotl => {
let n = ins.rot;
for lane in 0..LANES {
r[d][lane] = r[d][lane].rotate_left(n);
}
}
Op::Rotr => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].rotate_right(src[lane] & 31);
}
}
Op::Mad => {
let src = r[a];
let src2 = r[ins.src2 as usize];
for lane in 0..LANES {
r[d][lane] = src[lane].wrapping_mul(src2[lane]).wrapping_add(r[d][lane]);
}
}
Op::Shfl => {
let src = r[a];
let m = ins.mask as usize;
for lane in 0..LANES {
r[d][lane] ^= src[lane ^ m];
}
}
Op::Load => {
assert_eq!(ins.width, 1, "class v4 loads one word");
let mut ts = [0u32; LANES];
let mut ws = [0u32; LANES];
let p = sink.items.len();
for lane in 0..LANES {
let x = r[a][lane];
if x == 0 || x == u32::MAX {
sink.sat[p] += 1;
}
let idx = load_index(self.era, ins, x, self.mask, self.log2);
let (t, j) = self.layout.split(idx);
let t = if self.plant_site == Some(*site) { 0x00_1234 } else { t };
ts[lane] = t;
ws[lane] = idx;
r[d][lane] ^= self.table.items[t as usize][j as usize];
}
sink.items.push(ts);
sink.addrs.push(ws);
*site += 1;
}
Op::WLoad | Op::Scratch | Op::Hot => panic!("op {:?} is not a class v4 op", ins.op),
}
}
/// Hashes of the warp at `base_nonce`; the sink carries the item index of every load (128 x 32).
fn warp(&self, base_nonce: u32, sink: &mut Sink) -> [u64; LANES] {
let seed = &self.program.seed;
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
let nonce = base_nonce.wrapping_add(lane as u32);
for i in 0..8 {
let mut x = nonce ^ seed[i];
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
x = splitmix32(x);
r[i][lane] = x ^ seed[(i + 1) & 7];
}
}
sink.items.clear();
sink.addrs.clear();
sink.sat = [0; POS];
for _ in 0..ITERATIONS {
let sel = r[0];
let mut site = 0usize;
for ins in &self.program.instrs {
self.step(ins, &mut r, &sel, sink, &mut site);
}
for _ in 0..self.program.shadow_reps() {
for ins in &self.program.shadow {
self.step(ins, &mut r, &sel, sink, &mut site);
}
}
}
let mut hashes = [0u64; LANES];
for lane in 0..LANES {
let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21);
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
hashes[lane] = ((hi as u64) << 32) | lo as u64;
}
hashes
}
}
fn pct(h: &[u64], q: f64) -> usize {
let total: u64 = h.iter().sum();
let want = (total as f64 * q).ceil().max(1.0) as u64;
let mut acc = 0u64;
for (i, c) in h.iter().enumerate() {
acc += c;
if acc >= want {
return i;
}
}
h.len() - 1
}
fn dist_line(label: &str, h: &[u64]) -> String {
let total: u64 = h.iter().sum();
let min = h.iter().position(|&c| c > 0).unwrap_or(0);
let max = h.iter().rposition(|&c| c > 0).unwrap_or(0);
let mean = h.iter().enumerate().map(|(i, c)| i as f64 * *c as f64).sum::<f64>() / total.max(1) as f64;
format!(
"{label}: n {total} min {min} p1 {} median {} p99 {} max {max} mean {mean:.4}",
pct(h, 0.01),
pct(h, 0.5),
pct(h, 0.99)
)
}
/// The window of every load site as an item range: `(first item, items)`; the era window is the top `k` bits of
/// the 28-bit word index and every layout position lies below 16, so the window is the same aligned range of items.
fn site_item_windows(program: &Program, mask: u32, log2: u32) -> Vec<(u32, u32)> {
program
.instrs
.iter()
.filter(|i| i.op == Op::Load)
.map(|i| {
let (wm, off) = igneum_pow::verify::window(i, mask, log2);
(off >> 4, (wm >> 4) + 1)
})
.collect()
}
/// The expected reads per item under the window layer alone (every site uniform on its own window), as a density per
/// quarter of the item space (windows are the dataset, a half or a quarter, aligned).
fn window_density(windows: &[(u32, u32)], items_n: usize, reads_per_site: f64) -> [f64; 4] {
let q = items_n as u32 / 4;
let mut d = [0f64; 4];
for &(first, n) in windows {
for (k, dq) in d.iter_mut().enumerate() {
let qs = k as u32 * q;
if qs >= first && qs < first + n {
*dq += reads_per_site / n as f64;
}
}
}
d
}
/// Statistics of `counts` against a per-quarter expected density: chi-square per dof, the largest and smallest
/// bucket in sigma of their own expectation, buckets of `per` items.
fn stats_vs_density(counts: &[u64], dens: &[f64; 4], per: usize) -> (f64, f64, u64, usize, f64) {
let items_n = counts.len();
let q = items_n / 4;
let mut chi2 = 0f64;
let mut z_max = f64::MIN;
let mut z_min = f64::MAX;
let mut max = 0u64;
let mut argmax = 0usize;
let mut dof = 0usize;
for (b, c) in counts.chunks(per).enumerate() {
let e = dens[(b * per) / q] * per as f64;
if e <= 0.0 {
continue;
}
let s: u64 = c.iter().sum();
let z = (s as f64 - e) / e.sqrt();
chi2 += z * z;
dof += 1;
if z > z_max {
z_max = z;
max = s;
argmax = b;
}
if z < z_min {
z_min = z;
}
}
(chi2 / (dof.max(2) - 1) as f64, z_max, max, argmax, z_min)
}
/// The windowed control: `n` reads, site `i mod 16`, uniform on the site's window.
fn windowed_control(windows: &[(u32, u32)], n: u64, seed: u64, threads: usize, hist: &[AtomicU32]) {
std::thread::scope(|sc| {
for th in 0..threads {
let hist = &hist;
sc.spawn(move || {
let mut s = SplitMix64::new(seed ^ (th as u64).wrapping_mul(0x9E3779B97F4A7C15));
let per = n / threads as u64 + if (th as u64) < n % threads as u64 { 1 } else { 0 };
for i in 0..per {
let (first, cnt) = windows[(i % 16) as usize];
let t = first + (s.next() as u32 & (cnt - 1));
hist[t as usize].fetch_add(1, Ordering::Relaxed);
}
});
}
});
}
struct ProgramSummary {
label: String,
id: u64,
attempt: u32,
/// X_f against the windowed control at f = 0.1%, 0.5%, 1%
x: [f64; 3],
hot: bool,
/// windowed chi2/dof of the 64-item buckets and the largest bucket in sigma
chi2_w: f64,
z_w: f64,
/// the largest share of one hi16 bucket (256 items) at any load position
max_bucket_share: f64,
/// the largest saturated-source share at any load position
max_sat_share: f64,
/// acceptance-style 2,048-evaluation metrics: the largest count of one address at one position, the largest
/// saturated-source count at one position
acc_max_addr: u32,
acc_max_sat: u32,
/// S_0.1% over the window-model control and over the flat control
ratio_w: f64,
ratio_flat: f64,
/// top-f shares measured and under the window-model control, f = 0.1% and 1%
s01: f64,
e01: f64,
s10: f64,
e10: f64,
/// the hottest item, its count, and the lossy site whose saturated (or one-bit-off) source maps to it, if any
hottest: u32,
hottest_count: u64,
hottest_source: String,
/// share of each site's reads that land on the top 0.1% of items (the attribution pass; zeros without --diag)
by_site_share: [f64; 16],
}
#[allow(clippy::too_many_arguments)]
fn run_program(
spec: &ProgramSpec,
ds: DatasetSource,
table: &Table,
table_mism: u64,
day: u64,
nonces: u64,
threads: usize,
plant: Plant,
check_every: u64,
diag: bool,
verbose: bool,
out_dir: &str,
) -> (ProgramSummary, DatasetSource) {
let program = Epoch::chain_program(&spec.epoch_seed, Some(&spec.era), ProgramClass::V4, &spec.label);
assert_eq!(program.generator, 4);
assert_eq!(program.class.mixer_mult, 8);
assert_eq!(program.class.shadow.map(|s| (s.instrs, s.reps)), Some((256, 27)));
let era = program.class.era.expect("class v4 draws the era");
let n_loads = program.instrs.iter().filter(|i| i.op == Op::Load).count();
assert_eq!(n_loads, 16);
log!(
"program {}: epoch seed {} era seed {} id {:016x} attempt {} class {} op mix {}; era stride mul {:#010x} rot {} interleave {:?} windows {}",
spec.label,
hex(&spec.epoch_seed),
hex(&spec.era),
program.program_id(),
program.attempt,
program.class.name(),
program.op_mix(),
era.stride_mul,
era.stride_rot,
era.pos,
program.instrs.iter().enumerate().filter(|(_, i)| i.op == Op::Load).map(|(k, i)| format!("{k}:{}:{}", i.win, i.off)).collect::<Vec<_>>().join(" ")
);
let sites: Vec<usize> = program.instrs.iter().enumerate().filter(|(_, i)| i.op == Op::Load).map(|(k, _)| k).collect();
if verbose {
// the static shape of every load site: its source register, the last base-program writer of that register
// before the site (cyclic) and whether that writer injects (add, sub, xor, mad, shfl, load), and how often the
// shadow block writes the register (the shadow runs between iteration i's instruction 63 and iteration i+1's
// instruction 0, outside the acceptance rule)
let instrs = &program.instrs;
for (k, ins) in instrs.iter().enumerate() {
if ins.op != Op::Load {
continue;
}
let src = ins.src;
let mut writer = String::from("none");
let mut chain = Vec::new();
for back in 1..instrs.len() {
let j = (k + instrs.len() - back) % instrs.len();
if instrs[j].dst == src {
if writer == "none" {
writer = format!("{} at {j}", instrs[j].op.name());
}
chain.push(format!("{}@{j}", instrs[j].op.name()));
if instrs[j].op.injects() {
break;
}
}
}
let mut sh: std::collections::BTreeMap<&str, usize> = std::collections::BTreeMap::new();
for s in program.shadow.iter().filter(|s| s.dst == src) {
*sh.entry(s.op.name()).or_insert(0) += 1;
}
log!(
"load site instr {k}: src r{src} win {} off {}; last base writer {writer}; writers back to the last injecting one: {}; shadow writes of r{src} per rep: {}",
ins.win,
ins.off,
chain.join(" "),
sh.iter().map(|(o, n)| format!("{o}={n}")).collect::<Vec<_>>().join(" ")
);
}
}
let epoch = Epoch { program, dataset: ds };
let plant_site = if plant == Plant::ConstItem { Some(0) } else { None };
let mirror = Mirror {
program: &epoch.program,
era: epoch.program.class.era.as_ref(),
layout: epoch.program.class.layout(),
mask: epoch.dataset.mask,
log2: epoch.dataset.log2_words,
table,
plant_site,
};
let items_n = 1usize << (DATASET_LOG2 - 4);
let windows = site_item_windows(&epoch.program, epoch.dataset.mask, epoch.dataset.log2_words);
let dens = window_density(&windows, items_n, nonces as f64 * 8.0);
log!(
"window layer: site item windows (first, items) {}; expected reads per item by quarter {:.3} {:.3} {:.3} {:.3} (flat uniform {:.3})",
windows.iter().map(|(f, n)| format!("({:#x},2^{})", f, n.trailing_zeros())).collect::<Vec<_>>().join(" "),
dens[0],
dens[1],
dens[2],
dens[3],
nonces as f64 * 128.0 / items_n as f64
);
let item_hist = atomic_vec(items_n);
let pos_hi: Vec<Vec<AtomicU32>> = (0..POS).map(|_| atomic_vec(1 << 16)).collect();
let pos_sat: Vec<AtomicU32> = atomic_vec(POS);
let warps = nonces / LANES as u64;
let next = AtomicUsize::new(0);
let checked = AtomicUsize::new(0);
let hash_mism = AtomicUsize::new(0);
let max_lines_hash = 8 * 128;
let max_lines_warp = max_lines_hash * LANES;
// the acceptance-style metric on the first 64 warps (2,048 evaluations): per position, the count of every
// masked address and of saturated sources
let acc = std::sync::Mutex::new((vec![std::collections::HashMap::<u32, u32>::new(); POS], [0u32; POS]));
let t1 = Instant::now();
let dists: Vec<(Vec<u64>, Vec<u64>, Vec<u64>, Vec<u64>)> = std::thread::scope(|sc| {
let mut hs = Vec::new();
for _ in 0..threads {
let (next, checked, hash_mism, item_hist, mirror, epoch, pos_hi, pos_sat, acc) =
(&next, &checked, &hash_mism, &item_hist, &mirror, &epoch, &pos_hi, &pos_sat, &acc);
hs.push(sc.spawn(move || {
let mut lines_hash = vec![0u64; max_lines_hash + 1];
let mut items_hash = vec![0u64; 129];
let mut lines_warp = vec![0u64; max_lines_warp + 1];
let mut items_warp = vec![0u64; 128 * LANES + 1];
let mut sink = Sink::new();
let mut lane_lines: Vec<u32> = Vec::with_capacity(max_lines_hash);
let mut lane_items: Vec<u32> = Vec::with_capacity(128);
let mut warp_lines: Vec<u32> = Vec::with_capacity(max_lines_warp);
let mut warp_items: Vec<u32> = Vec::with_capacity(128 * LANES);
loop {
let w = next.fetch_add(1, Ordering::Relaxed) as u64;
if w >= warps {
break;
}
let base = (w * LANES as u64) as u32;
let hashes = mirror.warp(base, &mut sink);
assert_eq!(sink.items.len(), POS);
if plant == Plant::None && (w < 64 || w % check_every == 0) {
let lib = epoch.hash_warp(base);
checked.fetch_add(1, Ordering::Relaxed);
if lib != hashes {
hash_mism.fetch_add(1, Ordering::Relaxed);
}
}
if w < 64 {
let mut g = acc.lock().unwrap();
for p in 0..POS {
for lane in 0..LANES {
*g.0[p].entry(sink.addrs[p][lane]).or_insert(0) += 1;
}
g.1[p] += sink.sat[p];
}
}
warp_lines.clear();
warp_items.clear();
for lane in 0..LANES {
lane_lines.clear();
lane_items.clear();
for load in &sink.items {
let t = load[lane];
lane_items.push(t);
lane_lines.extend_from_slice(&mirror.table.lines[t as usize]);
}
warp_items.extend_from_slice(&lane_items);
warp_lines.extend_from_slice(&lane_lines);
lane_items.sort_unstable();
lane_items.dedup();
lane_lines.sort_unstable();
lane_lines.dedup();
items_hash[lane_items.len()] += 1;
lines_hash[lane_lines.len()] += 1;
}
for (p, load) in sink.items.iter().enumerate() {
for lane in 0..LANES {
let t = load[lane];
item_hist[t as usize].fetch_add(1, Ordering::Relaxed);
pos_hi[p][(t >> 8) as usize].fetch_add(1, Ordering::Relaxed);
}
pos_sat[p].fetch_add(sink.sat[p], Ordering::Relaxed);
}
warp_items.sort_unstable();
warp_items.dedup();
warp_lines.sort_unstable();
warp_lines.dedup();
items_warp[warp_items.len()] += 1;
lines_warp[warp_lines.len()] += 1;
}
(lines_hash, items_hash, lines_warp, items_warp)
}));
}
hs.into_iter().map(|h| h.join().unwrap()).collect()
});
let el = t1.elapsed().as_secs_f64();
let mut lines_hash = vec![0u64; max_lines_hash + 1];
let mut items_hash = vec![0u64; 129];
let mut lines_warp = vec![0u64; max_lines_warp + 1];
let mut items_warp = vec![0u64; 128 * LANES + 1];
for (a, b, c, d) in &dists {
for (i, v) in a.iter().enumerate() {
lines_hash[i] += v;
}
for (i, v) in b.iter().enumerate() {
items_hash[i] += v;
}
for (i, v) in c.iter().enumerate() {
lines_warp[i] += v;
}
for (i, v) in d.iter().enumerate() {
items_warp[i] += v;
}
}
let ck = checked.load(Ordering::Relaxed);
let hm = hash_mism.load(Ordering::Relaxed);
log!(
"{} warps ({} nonces) interpreted in {el:.1} s ({:.3} ms per warp per thread); Epoch::hash_warp agreement on {ck} warps: {hm} mismatches{}",
warps,
warps * LANES as u64,
el * 1e3 * threads as f64 / warps as f64,
if plant != Plant::None { " (library comparison skipped under the plant)" } else { "" }
);
if hm > 0 || table_mism > 0 {
log!("THE MIRROR DISAGREES WITH THE LIBRARY (hashes {hm}, items {table_mism}); nothing below is trusted");
}
let tag = format!("warps-{}-d{day}-n{nonces}-{}", spec.label, plant.name());
log!("{}", dist_line(&format!("{} distinct lines per hash", spec.label), &lines_hash));
log!("{}", dist_line(&format!("{} distinct items per hash", spec.label), &items_hash));
log!("{}", dist_line(&format!("{} distinct lines per warp", spec.label), &lines_warp));
log!("{}", dist_line(&format!("{} distinct items per warp", spec.label), &items_warp));
let exp = |k: f64, l: f64| l * (1.0 - (1.0 - 1.0 / l).powf(k));
log!(
"uniform expectation: distinct lines per hash {:.3} of 1024 reads, per warp {:.1} of 32768 reads (2^22 lines); distinct items per hash {:.4} of 128, per warp {:.2} of 4096 (2^24 items)",
exp(1024.0, 4194304.0),
exp(32768.0, 4194304.0),
exp(128.0, 16777216.0),
exp(4096.0, 16777216.0)
);
{
let mut f = std::fs::File::create(format!("{out_dir}/{tag}-distinct.txt")).unwrap();
for (name, h) in [("lines_hash", &lines_hash), ("items_hash", &items_hash), ("lines_warp", &lines_warp), ("items_warp", &items_warp)] {
writeln!(f, "# distinct {name}: value count").unwrap();
for (i, c) in h.iter().enumerate().filter(|(_, c)| **c > 0) {
writeln!(f, "{name} {i} {c}").unwrap();
}
}
}
// per-position diagnostics: the largest hi16 bucket share and the saturated-source share
let mut max_bucket_share = 0f64;
let mut max_bucket_pos = 0usize;
let mut max_sat_share = 0f64;
let mut max_sat_pos = 0usize;
{
let mut f = std::fs::File::create(format!("{out_dir}/{tag}-positions.txt")).unwrap();
writeln!(f, "# position iteration site instr max_hi16_bucket_share saturated_share").unwrap();
for p in 0..POS {
let mx = pos_hi[p].iter().map(|x| x.load(Ordering::Relaxed)).max().unwrap_or(0) as f64 / nonces as f64;
let sat = pos_sat[p].load(Ordering::Relaxed) as f64 / nonces as f64;
writeln!(f, "{p} {} {} {} {:.6} {:.6}", p / 16, p % 16, sites[p % 16], mx, sat).unwrap();
if mx > max_bucket_share {
max_bucket_share = mx;
max_bucket_pos = p;
}
if sat > max_sat_share {
max_sat_share = sat;
max_sat_pos = p;
}
}
}
log!(
"positions: largest hi16-bucket (256 items) share {:.4}% at p{} (iteration {}, site {}, instr {}); windowed expectation {:.4}%; largest saturated-source share {:.4}% at p{} (site {}, instr {})",
max_bucket_share * 100.0,
max_bucket_pos,
max_bucket_pos / 16,
max_bucket_pos % 16,
sites[max_bucket_pos % 16],
100.0 * 256.0 / (windows[max_bucket_pos % 16].1 as f64),
max_sat_share * 100.0,
max_sat_pos,
max_sat_pos % 16,
sites[max_sat_pos % 16]
);
// the acceptance-style metric over the first 2,048 evaluations
let (acc_max_addr, acc_max_sat, acc_pos, acc_sat_pos) = {
let g = acc.lock().unwrap();
let mut ma = 0u32;
let mut mp = 0usize;
let mut ms = 0u32;
let mut msp = 0usize;
for p in 0..POS {
let m = g.0[p].values().copied().max().unwrap_or(0);
if m > ma {
ma = m;
mp = p;
}
if g.1[p] > ms {
ms = g.1[p];
msp = p;
}
}
(ma, ms, mp, msp)
};
log!(
"acceptance-style (2,048 evaluations, the rule's sample size): the most repeated address at one position {} of 2048 at p{} (site {}, instr {}); saturated sources at one position {} of 2048 at p{} (site {}, instr {}); uniform expectation: repeats 1 to 2, saturated 0",
acc_max_addr,
acc_pos,
acc_pos % 16,
sites[acc_pos % 16],
acc_max_sat,
acc_sat_pos,
acc_sat_pos % 16,
sites[acc_sat_pos % 16]
);
// the cross-hash item histogram: flat statistics, the windowed null, the windowed control, the hot-set test
let snap = snapshot(&item_hist);
let s_items = stats_of(snap.iter().copied(), items_n);
log!("{}", s_items.line(&format!("{} item histogram (flat)", spec.label)));
let (chi2_w1, z_w1, max_w1, arg_w1, zmin_w1) = stats_vs_density(&snap, &dens, 1);
let (chi2_w, z_w, max_w, arg_w, zmin_w) = stats_vs_density(&snap, &dens, 64);
log!(
"{} item histogram against the window density: full 2^24 chi2/dof {:.5} largest {} (item {:#x}) at {:+.2} sigma smallest at {:+.2}; buckets64 chi2/dof {:.5} largest {} (bucket {}) at {:+.2} sigma smallest at {:+.2}",
spec.label,
chi2_w1,
max_w1,
arg_w1,
z_w1,
zmin_w1,
chi2_w,
max_w,
arg_w,
z_w,
zmin_w
);
let total_reads = nonces * 128;
let ctrl = atomic_vec(items_n);
windowed_control(&windows, total_reads, 0xF8_3333 ^ program_hash(&spec.label), threads, &ctrl);
let cs = snapshot(&ctrl);
let c_items = stats_of(cs.iter().copied(), items_n);
let (c_chi2_w, c_z_w, _, _, c_zmin_w) = stats_vs_density(&cs, &dens, 64);
let (c_chi2_w1, c_z_w1, _, _, _) = stats_vs_density(&cs, &dens, 1);
log!("{}", c_items.line(&format!("{} WINDOWED CONTROL item histogram (flat)", spec.label)));
log!(
"{} WINDOWED CONTROL against the window density: full chi2/dof {:.5} largest {:+.2} sigma; buckets64 chi2/dof {:.5} largest {:+.2} smallest {:+.2}",
spec.label,
c_chi2_w1,
c_z_w1,
c_chi2_w,
c_z_w,
c_zmin_w
);
let f1 = z_w > 6.0 || zmin_w < -6.0;
log!(
"6-sigma {} items buckets64 against the window density: largest bucket {:+.2} sigma, smallest {:+.2} -> {}",
spec.label,
z_w,
zmin_w,
if f1 { "FLAGGED (beyond 6 sigma)" } else { "within 6 sigma" }
);
let f3 = hot_set_test(&format!("{} items (windowed control)", spec.label), &s_items, &c_items);
let flat = atomic_vec(items_n);
control(items_n, total_reads, 0xF8_4444 ^ program_hash(&spec.label), threads, &flat);
let fs = snapshot(&flat);
let f_items = stats_of(fs.iter().copied(), items_n);
log!("{}", f_items.line(&format!("{} FLAT CONTROL item histogram", spec.label)));
let _ = hot_set_test(&format!("{} items (flat control, the auditor's first view)", spec.label), &s_items, &f_items);
let mut x = [0f64; 3];
let mut ratio_w = [0f64; 3];
let mut ratio_flat = [0f64; 3];
for k in 0..3 {
x[k] = s_items.top[k].1 - c_items.top[k].1;
ratio_w[k] = s_items.top[k].1 / c_items.top[k].1;
ratio_flat[k] = s_items.top[k].1 / f_items.top[k].1;
}
log!(
"ratio {}: top 0.1% / 0.5% / 1% share over the WINDOW-MODEL control {:.4}x / {:.4}x / {:.4}x (gate 1.2x at 0.1%: {}); over the FLAT control {:.4}x / {:.4}x / {:.4}x",
spec.label,
ratio_w[0],
ratio_w[1],
ratio_w[2],
if ratio_w[0] <= 1.2 { "within" } else { "BEYOND" },
ratio_flat[0],
ratio_flat[1],
ratio_flat[2]
);
{
let mut g = std::fs::File::create(format!("{out_dir}/{tag}-items.u32le")).unwrap();
let mut bytes = Vec::with_capacity(items_n * 4);
for c in &snap {
bytes.extend_from_slice(&(*c as u32).to_le_bytes());
}
g.write_all(&bytes).unwrap();
}
let mut by_site_share = [0f64; 16];
if diag {
// attribution: the hottest items (the top 0.1% by count, and the top 8) traced back to the load positions
let want = (items_n as f64 / 1000.0).round() as u64;
let mut coc: std::collections::BTreeMap<u64, u64> = std::collections::BTreeMap::new();
for &c in &snap {
*coc.entry(c).or_insert(0) += 1;
}
let mut left = want;
let mut thr = 0u64;
for (&c, &n) in coc.iter().rev() {
thr = c;
if n >= left {
break;
}
left -= n;
}
let mut hot = vec![0u8; items_n];
let mut n_hot = 0u64;
let mut hot_reads = 0u64;
for (t, &c) in snap.iter().enumerate() {
if c >= thr {
hot[t] = 1;
n_hot += 1;
hot_reads += c;
}
}
let mut top8: Vec<(u64, usize)> = snap.iter().enumerate().map(|(t, &c)| (c, t)).collect();
top8.sort_unstable_by(|a, b| b.cmp(a));
top8.truncate(8);
log!(
"attribution: hot threshold count >= {thr} marks {n_hot} items ({:.4}% of items) holding {hot_reads} reads ({:.4}% of reads); top 8 items {}",
n_hot as f64 * 100.0 / items_n as f64,
hot_reads as f64 * 100.0 / total_reads as f64,
top8.iter().map(|(c, t)| format!("t={t:#08x}:{c}")).collect::<Vec<_>>().join(" ")
);
let per_pos: Vec<std::sync::atomic::AtomicU64> = (0..POS).map(|_| std::sync::atomic::AtomicU64::new(0)).collect();
let per_top: Vec<Vec<std::sync::atomic::AtomicU64>> = (0..8).map(|_| (0..POS).map(|_| std::sync::atomic::AtomicU64::new(0)).collect()).collect();
let next2 = AtomicUsize::new(0);
std::thread::scope(|sc| {
for _ in 0..threads {
let (next2, mirror, hot, top8, per_pos, per_top) = (&next2, &mirror, &hot, &top8, &per_pos, &per_top);
sc.spawn(move || {
let mut sink = Sink::new();
loop {
let w = next2.fetch_add(1, Ordering::Relaxed) as u64;
if w >= warps {
break;
}
mirror.warp((w * LANES as u64) as u32, &mut sink);
for (p, load) in sink.items.iter().enumerate() {
for lane in 0..LANES {
let t = load[lane] as usize;
if hot[t] != 0 {
per_pos[p].fetch_add(1, Ordering::Relaxed);
}
for (k, (_, tt)) in top8.iter().enumerate() {
if *tt == t {
per_top[k][p].fetch_add(1, Ordering::Relaxed);
}
}
}
}
}
});
}
});
let expect = n_hot as f64 / items_n as f64;
let mut rows: Vec<(usize, u64)> = per_pos.iter().enumerate().map(|(p, c)| (p, c.load(Ordering::Relaxed))).collect();
rows.sort_unstable_by(|a, b| b.1.cmp(&a.1));
log!(
"attribution: reads on hot items per position (flat expectation {:.4}% of each position's {nonces} reads); the 24 largest: {}",
expect * 100.0,
rows.iter().take(24).map(|(p, c)| format!("p{p}(it{} s{})={:.3}%", p / 16, p % 16, *c as f64 * 100.0 / nonces as f64)).collect::<Vec<_>>().join(" ")
);
let mut by_iter = vec![0u64; ITERATIONS];
let mut by_site = vec![0u64; 16];
for (p, c) in &rows {
by_iter[p / 16] += c;
by_site[p % 16] += c;
}
for s in 0..16 {
by_site_share[s] = by_site[s] as f64 / (nonces * 8) as f64;
}
log!(
"attribution: hot reads by iteration {} ; by site {}",
by_iter.iter().enumerate().map(|(i, c)| format!("it{i}={:.3}%", *c as f64 * 100.0 / (nonces * 16) as f64)).collect::<Vec<_>>().join(" "),
by_site.iter().enumerate().map(|(i, c)| format!("s{i}={:.3}%", *c as f64 * 100.0 / (nonces * 8) as f64)).collect::<Vec<_>>().join(" ")
);
// per-site contribution table: window, share of the site's reads into the top 0.1% set, the entropy of the
// site's item index over nonces (hi16 buckets of 256 items: a window of 2^(24 - k_off) items is 2^(16 - k_off)
// buckets, so uniform on the window = 16 - k_off bits; the runs of 09:04 UTC printed 16 - 2 k_off by mistake),
// saturated-source share, largest bucket share
let load_instrs: Vec<&Instr> = epoch.program.instrs.iter().filter(|i| i.op == Op::Load).collect();
for s in 0..16 {
let mut acc = vec![0u64; 1 << 16];
let mut hot_s = 0u64;
let mut sat_s = 0u64;
for it in 0..ITERATIONS {
let p = it * 16 + s;
for (b, c) in pos_hi[p].iter().enumerate() {
acc[b] += c.load(Ordering::Relaxed) as u64;
}
hot_s += per_pos[p].load(Ordering::Relaxed);
sat_s += pos_sat[p].load(Ordering::Relaxed) as u64;
}
let n_s = (nonces * ITERATIONS as u64) as f64;
let mut h = 0f64;
let mut mx = 0u64;
for &c in &acc {
if c > 0 {
let pr = c as f64 / n_s;
h -= pr * pr.log2();
}
mx = mx.max(c);
}
let ins = load_instrs[s];
log!(
"site {s} (instr {}, src r{}, k_off {} offset {}, window 2^{} items): share of its reads into the top 0.1% {:.4}% (flat expectation {:.4}%); index entropy {:.3} bits of {} uniform-on-window; saturated source {:.4}%; largest 256-item bucket {:.4}% (window expectation {:.4}%)",
sites[s],
ins.src,
ins.win,
ins.off,
windows[s].1.trailing_zeros(),
hot_s as f64 * 100.0 / n_s,
expect * 100.0,
h,
16 - ins.win.min(2) as u32,
sat_s as f64 * 100.0 / n_s,
mx as f64 * 100.0 / n_s,
100.0 * 256.0 / windows[s].1 as f64
);
}
for (k, (c, t)) in top8.iter().enumerate() {
let mut pos: Vec<(usize, u64)> = per_top[k].iter().enumerate().map(|(p, x)| (p, x.load(Ordering::Relaxed))).filter(|(_, x)| *x > 0).collect();
pos.sort_unstable_by(|a, b| b.1.cmp(&a.1));
log!(
"attribution: item {t:#08x} ({c} reads) read from positions {}",
pos.iter().take(12).map(|(p, x)| format!("p{p}(it{} s{})x{x}", p / 16, p % 16)).collect::<Vec<_>>().join(" ")
);
}
}
// the hottest item and the lossy site whose saturated source maps to it under the era map (all ones, zero, or
// one bit off either), with that site's source register and its last base-program writer
let (hottest, hottest_count) = snap.iter().enumerate().map(|(t, &c)| (t as u32, c)).max_by_key(|&(_, c)| c).unwrap();
let hottest_source = {
let instrs = &epoch.program.instrs;
let layout = epoch.program.class.layout();
let era = epoch.program.class.era.as_ref();
let mut found = String::from("none");
for (k, ins) in instrs.iter().enumerate() {
if ins.op != Op::Load {
continue;
}
let img = |x: u32| layout.split(load_index(era, ins, x, epoch.dataset.mask, epoch.dataset.log2_words)).0;
let mut cands: Vec<(u32, &str)> = vec![(u32::MAX, "all-ones"), (0, "zero")];
for b in 0..32 {
cands.push((u32::MAX ^ (1 << b), "one-zero-bit"));
cands.push((1 << b, "one-one-bit"));
}
if let Some((_, what)) = cands.iter().find(|(x, _)| img(*x) == hottest) {
let mut writer = String::from("none");
for back in 1..instrs.len() {
let j = (k + instrs.len() - back) % instrs.len();
if instrs[j].dst == ins.src {
writer = format!("{}@{j}", instrs[j].op.name());
break;
}
}
found = format!("site-instr{k}:r{}:{what}:last-writer-{writer}", ins.src);
break;
}
}
found
};
log!(
"hottest item {} ({:#08x}): {} reads ({:.4}% of all); predicted source {}",
spec.label,
hottest,
hottest_count,
hottest_count as f64 * 100.0 / total_reads as f64,
hottest_source
);
let hot = f3;
log!(
"program {} DONE: {} nonces; 6-sigma (windowed) {}; hot-set {}; verdict {}",
spec.label,
nonces,
if f1 { "FLAGGED" } else { "clear" },
if hot { "FLAGGED" } else { "clear" },
if f1 || hot { "FLAGGED" } else { "PASS (no test fired)" }
);
let summary = ProgramSummary {
label: spec.label.clone(),
id: epoch.program.program_id(),
attempt: epoch.program.attempt,
x,
hot,
chi2_w,
z_w,
max_bucket_share,
max_sat_share,
acc_max_addr,
acc_max_sat,
ratio_w: ratio_w[0],
ratio_flat: ratio_flat[0],
s01: s_items.top[0].1,
e01: c_items.top[0].1,
s10: s_items.top[2].1,
e10: c_items.top[2].1,
hottest,
hottest_count,
hottest_source,
by_site_share,
};
(summary, epoch.dataset)
}
fn program_hash(label: &str) -> u64 {
igneum_pow::seed::fnv1a64(label.as_bytes())
}
#[allow(clippy::too_many_arguments)]
fn census_warps(programs: &[u32], day: u64, nonces: u64, threads: usize, plant: Plant, validate: &str, check_every: u64, diag: bool, out_dir: &str) {
log!(
"census warps: programs {:?} day {day} nonces {nonces} threads {threads} plant {} validate {validate} check_every {check_every} diag {diag}",
programs,
plant.name()
);
let t0 = Instant::now();
let mut ds: DatasetSource = Epoch::chain_dataset_day(&day_bytes(day), ProgramClass::V4, 0, DATASET_LOG2);
assert_eq!(ds.log2_words, DATASET_LOG2);
let (table, table_mism) = {
let mh = ds.memhard().unwrap();
log!("day {day}: cache filled in {:.2} s, fnv {:016x}", t0.elapsed().as_secs_f64(), mh.cache.fnv1a64());
build_table(&mh.params, &mh.cache, threads, plant, validate)
};
let verbose = programs.len() <= 3;
let mut summaries = Vec::new();
for &k in programs {
let spec = program_spec(k);
let (s, d) = run_program(&spec, ds, &table, table_mism, day, nonces, threads, plant, check_every, diag, verbose, out_dir);
ds = d;
summaries.push(s);
}
if programs.len() > 1 {
// the re-gate format (Counter ASIC lane, 7 October 2026): one line per seed, then PASS or FAIL against 1.2x
let mut over = Vec::new();
for s in &summaries {
log!(
"SEED {} id {:016x} attempt {} S0.1 {:.5}% null0.1 {:.5}% ratio {:.4}x S1.0 {:.5}% null1.0 {:.5}% hottest {:#08x}:{} source {} by-site {}",
s.label,
s.id,
s.attempt,
s.s01 * 100.0,
s.e01 * 100.0,
s.ratio_w,
s.s10 * 100.0,
s.e10 * 100.0,
s.hottest,
s.hottest_count,
s.hottest_source,
if diag { s.by_site_share.iter().enumerate().map(|(i, x)| format!("s{i}={:.3}%", x * 100.0)).collect::<Vec<_>>().join(" ") } else { "(no --diag)".to_string() }
);
if s.ratio_w > 1.2 {
over.push(format!("{}({:.3}x)", s.label, s.ratio_w));
}
}
log!(
"CENSUS {}: {} seeds at {} nonces, {} over 1.2x of the window model at the top 0.1%{}{}",
if over.is_empty() { "PASS" } else { "FAIL" },
summaries.len(),
nonces,
over.len(),
if over.is_empty() { "" } else { ": " },
over.join(" ")
);
let mut f = std::fs::File::create(format!("{out_dir}/seed-census-d{day}-n{nonces}-p{}-{}.txt", programs[0], programs[programs.len() - 1])).unwrap();
writeln!(f, "# label id attempt X_0.1% X_0.5% X_1% hot chi2w_buckets64 z_w max_hi16_bucket_share max_sat_share acc_max_addr_of_2048 acc_max_sat_of_2048 ratio_0.1%_window ratio_0.1%_flat").unwrap();
let mut n_hot = 0usize;
let mut n_acc_addr = [0usize; 3];
let mut n_acc_sat = [0usize; 3];
let mut n_sixsig = 0usize;
let mut n_ratio = 0usize;
for s in &summaries {
n_ratio += (s.ratio_w > 1.2) as usize;
writeln!(
f,
"{} {:016x} {} {:.5} {:.5} {:.5} {} {:.4} {:.2} {:.6} {:.6} {} {} {:.4} {:.4}",
s.label, s.id, s.attempt, s.x[0] * 100.0, s.x[1] * 100.0, s.x[2] * 100.0, s.hot as u8, s.chi2_w, s.z_w, s.max_bucket_share, s.max_sat_share, s.acc_max_addr, s.acc_max_sat, s.ratio_w, s.ratio_flat
)
.unwrap();
n_hot += s.hot as usize;
n_sixsig += (s.z_w > 6.0) as usize;
for (i, thr) in [3u32, 11, 21].into_iter().enumerate() {
n_acc_addr[i] += (s.acc_max_addr >= thr) as usize;
n_acc_sat[i] += (s.acc_max_sat >= thr) as usize;
}
}
log!(
"SEED CENSUS: {} programs at {} nonces each: hot set (X_f >= f, windowed control) in {}; top 0.1% beyond 1.2x of the window-model control in {}; windowed buckets64 beyond 6 sigma in {}; acceptance-style most-repeated address at one position >= 3 / 11 / 21 of 2048 in {} / {} / {}; saturated sources at one position >= 3 / 11 / 21 of 2048 in {} / {} / {}",
summaries.len(),
nonces,
n_hot,
n_ratio,
n_sixsig,
n_acc_addr[0],
n_acc_addr[1],
n_acc_addr[2],
n_acc_sat[0],
n_acc_sat[1],
n_acc_sat[2]
);
let mut by_x: Vec<&ProgramSummary> = summaries.iter().collect();
by_x.sort_by(|a, b| b.x[0].partial_cmp(&a.x[0]).unwrap());
for s in by_x.iter().take(10) {
log!(
"SEED CENSUS top: {} id {:016x} attempt {} ratio_0.1% window {:.3}x flat {:.3}x X_0.1% {:+.4}% X_1% {:+.4}% hot {} z_w {:+.1} max bucket share {:.4}% max sat share {:.4}% acc addr {} acc sat {}",
s.label,
s.id,
s.attempt,
s.ratio_w,
s.ratio_flat,
s.x[0] * 100.0,
s.x[2] * 100.0,
s.hot as u8,
s.z_w,
s.max_bucket_share * 100.0,
s.max_sat_share * 100.0,
s.acc_max_addr,
s.acc_max_sat
);
}
}
}
fn usage() -> ! {
eprintln!(
"attack-f8 lines --day0 20730 --days 16 --items-log2 24 --threads 12 --out <dir> [--plant none|quarter-lines|half-lines] [--validate all|sample|none]\n\
attack-f8 warps --program 1|2|3 | --programs a..b --day 20730 --nonces 1000000 --threads 12 --out <dir> [--plant none|const-item] [--validate all|sample|none] [--check-every 997] [--diag 1]\n\
attack-f8 census --seeds 64 --nonces 16777216 --control window --by-site --threads 12 --out <dir> the re-gate: one line per seed, PASS/FAIL against 1.2x\n\
attack-f8 static --programs 1..67 per program, every load site's write chain back to the last injecting write (rule (a'))\n\
attack-f8 seeds print the three programs' seeds"
);
std::process::exit(2);
}
fn main() {
let args: Vec<String> = std::env::args().collect();
if args.len() < 2 {
usage();
}
let get = |k: &str, d: &str| -> String {
let mut i = 2;
while i + 1 < args.len() {
if args[i] == k {
return args[i + 1].clone();
}
i += 1;
}
d.to_string()
};
let threads: usize = get("--threads", "12").parse().unwrap();
let out = get("--out", ".");
let plant = Plant::parse(&get("--plant", "none"));
let validate = get("--validate", "all");
log!("attack-f8 {} (igneum-pow {}); args {:?}", env!("CARGO_PKG_VERSION"), igneum_pow::generator::GENERATOR_VERSION_V4, &args[1..]);
match args[1].as_str() {
"lines" => census_lines(
get("--day0", &DEFAULT_DAY.to_string()).parse().unwrap(),
get("--days", "16").parse().unwrap(),
get("--items-log2", "24").parse().unwrap(),
threads,
plant,
&validate,
&out,
),
"warps" => census_warps(
&{
let ps = get("--programs", "");
if ps.is_empty() {
vec![get("--program", "1").parse::<u32>().unwrap()]
} else {
let (a, b) = ps.split_once("..").expect("--programs a..b");
(a.parse::<u32>().unwrap()..=b.parse::<u32>().unwrap()).collect::<Vec<u32>>()
}
},
get("--day", &DEFAULT_DAY.to_string()).parse().unwrap(),
get("--nonces", "1000000").parse().unwrap(),
threads,
plant,
&validate,
get("--check-every", "997").parse().unwrap(),
get("--diag", "0") == "1",
&out,
),
"census" => {
// the re-gate entry: `census --seeds 64 --nonces 16777216 --control window --by-site` runs programs 2..=65
// (the chain-shaped tag seeds; `--programs a..b` overrides) with the window-model control (the only control
// the hot-set test uses; the flat control is printed beside it) and the by-site attribution
let ps = get("--programs", "");
let programs: Vec<u32> = if ps.is_empty() {
let n: u32 = get("--seeds", "64").parse().unwrap();
(2..=n + 1).collect()
} else {
let (a, b) = ps.split_once("..").expect("--programs a..b");
(a.parse::<u32>().unwrap()..=b.parse::<u32>().unwrap()).collect()
};
let by_site = args.iter().any(|a| a == "--by-site") || get("--diag", "0") == "1";
census_warps(
&programs,
get("--day", &DEFAULT_DAY.to_string()).parse().unwrap(),
get("--nonces", "16777216").parse().unwrap(),
threads,
plant,
&validate,
get("--check-every", "997").parse().unwrap(),
by_site,
&out,
);
}
"static" => {
// per program: for every load site, the ops that write its source register between the register's last
// injecting write (add, sub, xor, mad, shfl, load) and the load, walking back cyclically; the proposed
// rule (a') rejects a program with an `or` or a `mul` in any such chain
let ps = get("--programs", "1..3");
let (a, b) = ps.split_once("..").expect("--programs a..b");
println!("# label id attempt sites_with_or sites_with_mul sites_with_mulhi sites_with_rot_only sites_clean rule_a_prime chains");
for k in a.parse::<u32>().unwrap()..=b.parse::<u32>().unwrap() {
let spec = program_spec(k);
let program = Epoch::chain_program(&spec.epoch_seed, Some(&spec.era), ProgramClass::V4, &spec.label);
let instrs = &program.instrs;
let (mut n_or, mut n_mul, mut n_mulhi, mut n_rot, mut n_clean) = (0, 0, 0, 0, 0);
let mut chains = Vec::new();
for (k, ins) in instrs.iter().enumerate() {
if ins.op != Op::Load {
continue;
}
let mut ops = Vec::new();
for back in 1..instrs.len() {
let j = (k + instrs.len() - back) % instrs.len();
if instrs[j].dst == ins.src {
if instrs[j].op.injects() {
ops.push(format!("{}@{j}", instrs[j].op.name()));
break;
}
ops.push(format!("{}@{j}", instrs[j].op.name()));
}
}
let has = |o: Op| ops.iter().any(|x| x.starts_with(&format!("{}@", o.name())));
if has(Op::Or) {
n_or += 1;
} else if has(Op::Mul) {
n_mul += 1;
} else if has(Op::MulHi) {
n_mulhi += 1;
} else if has(Op::Rotl) || has(Op::Rotr) {
n_rot += 1;
} else {
n_clean += 1;
}
// the items the saturated sources map to under the era map at the 2^28-word dataset
let mask = (1u32 << DATASET_LOG2) - 1;
let layout = program.class.layout();
let img = |x: u32| layout.split(load_index(program.class.era.as_ref(), ins, x, mask, DATASET_LOG2)).0;
chains.push(format!("{k}:r{}:{}:ones={:#08x}:zero={:#08x}", ins.src, ops.join("<"), img(u32::MAX), img(0)));
}
println!(
"{} {:016x} {} {n_or} {n_mul} {n_mulhi} {n_rot} {n_clean} {} {}",
spec.label,
program.program_id(),
program.attempt,
if n_or + n_mul > 0 { "REJECT" } else { "accept" },
chains.join(" ")
);
}
}
"seeds" => {
for k in 1..=3 {
let s = program_spec(k);
println!("{} epoch_seed {} era {}", s.label, hex(&s.epoch_seed), hex(&s.era));
}
}
_ => usage(),
}
}