From 2f718001e63e44d9b167f61408cda58bf6d33b32 Mon Sep 17 00:00:00 2001 From: igneum-josh <337424239+igneum-josh@users.noreply.github.com> Date: Wed, 7 Oct 2026 23:24:02 +0100 Subject: [PATCH] Counter ASIC 4.0 research: the per-load shadow's duplicate reads, found and fixed (the acceptance test judges the order the class executes) The finding (tests/ca4_trace.rs, the first export traced at 22:16 UTC): the per-load class derived 10,728 distinct items of 12,288 over three units (the class v4 shape 12,286), 1,482 same-iteration duplicate lanes at sites 8, 10 and 15; the mechanism is not the sub-block's last writer alone: a lossy base writer (mul, mulhi, or) followed by 27 passes of a 16-instruction map collapses a load's source register to 1 to 17 distinct values in 32 lanes before the next load (the census's 29 failing seeds of 64 at 22:16 UTC name mulhi, mul, or, rotl and load as the last base writers). The fix, two layers: (1) generator: a per-load sub-block instruction that writes the NEXT load's source register is redrawn from the injecting families when its op is mul, mulhi or or (consumes a draw; the class's own stream); (2) accept.rs: the dynamic test steps the per-load sub-blocks inside run_unit (alu_step, the same arithmetic as the arms) so saturation and distinctness are judged on the register file the loads read from, and a new rejection DuplicateLanes (the per-load class only) refuses a candidate whose load reads one address in two lanes of a unit; a rejected candidate redraws the attempt. After: the trace reads 12,287 of 12,288 with 0 duplicate lanes on the genesis seed; the 64-seed census on a second dataset reads 1 duplicate pair in 16,384 rows (seed ca4-census/49, the chance floor of a 2^24 index space, about 0.5 expected; the class v4 shape's own 2 of 12,288 are the same floor). verify.rs gains trace_load_indices (the diagnostic). Suite on igneum-build-2: 64 + 2 + 7 + 4 + 19 + 2 + 7 passed, 0 failed (RESULT rc=0, 22:23 UTC). The pinned packs unchanged; mx8+shl256x27's accepted attempt and id move, so the pack is re-exported. Co-Authored-By: Claude Fable 5.1 --- igneum-pow/src/accept.rs | 104 ++++++++++++++++++++++++++++++ igneum-pow/src/generator.rs | 40 ++++++++++-- igneum-pow/src/verify.rs | 56 +++++++++++++++++ igneum-pow/tests/ca4_trace.rs | 115 ++++++++++++++++++++++++++++++++++ 4 files changed, 311 insertions(+), 4 deletions(-) create mode 100644 igneum-pow/tests/ca4_trace.rs diff --git a/igneum-pow/src/accept.rs b/igneum-pow/src/accept.rs index 198871a28..ea7494a4e 100644 --- a/igneum-pow/src/accept.rs +++ b/igneum-pow/src/accept.rs @@ -61,6 +61,9 @@ pub enum Reject { OutputBias { bit: u8, ones: u32 }, /// (c): the distinct-address sum was `sum`. DistinctAddresses { sum: u64 }, + /// (c, the per-load shadow class only; Counter ASIC 4.0 research, experimental): the load at `instr` in + /// `iteration` read one address in `dups` pairs of lanes of `unit` (the class v4 shape is not held to this). + DuplicateLanes { iteration: u8, instr: u8, unit: u8, dups: u8 }, } impl std::fmt::Display for Reject { @@ -71,6 +74,9 @@ impl std::fmt::Display for Reject { } Reject::NoInjectingWrite { reg } => write!(f, "(b) r{reg} has no add, sub, xor, mad, shfl or load write"), Reject::ConstantBit { reg, bits } => write!(f, "(c) r{reg} has {bits} nonce-independent bits"), + Reject::DuplicateLanes { iteration, instr, unit, dups } => { + write!(f, "(c, per-load shadow) load at iteration {iteration} instruction {instr} reads a duplicate address in {dups} lanes of unit {unit}") + } Reject::LaneConstantSite { iteration, instr, unit } => { write!(f, "(c) load at iteration {iteration} instruction {instr} reads one address in all lanes of unit {unit}") } @@ -174,6 +180,82 @@ struct Acc { distinct_sum: u64, } +/// One ALU instruction of a shadow sub-block on the register file (the same arithmetic as the arms of `run_unit`; +/// no load, scratch or hot op is ever in a sub-block). Counter ASIC 4.0 research: the per-load shadow class runs its +/// sub-blocks inside the acceptance test, so the test judges the order the class executes. +fn alu_step(ins: &Instr, r: &mut [[u32; LANES]; 8], sel: &[u32; LANES]) { + let d = ins.dst as usize; + let a = ins.src as usize; + match ins.op { + Op::Add => { + let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32); + let src = r[a]; + for lane in 0..LANES { + let s = (sel[lane] >> bit) & 1; + let c = if s != 0 { imm2 } else { imm }; + r[d][lane] = r[d][lane].wrapping_add(src[lane]).wrapping_add(c); + } + } + Op::Sub => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].wrapping_sub(src[lane]); + } + } + Op::Mul => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].wrapping_mul(src[lane]); + } + } + Op::MulHi => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = mulhi32(r[d][lane], src[lane]); + } + } + Op::Xor => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] ^= src[lane]; + } + } + Op::Or => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] |= src[lane]; + } + } + Op::Rotl => { + let n = ins.rot; + for lane in 0..LANES { + r[d][lane] = r[d][lane].rotate_left(n); + } + } + Op::Rotr => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].rotate_right(src[lane] & 31); + } + } + Op::Mad => { + let src = r[a]; + let src2 = r[ins.src2 as usize]; + for lane in 0..LANES { + r[d][lane] = src[lane].wrapping_mul(src2[lane]).wrapping_add(r[d][lane]); + } + } + Op::Shfl => { + let src = r[a]; + let m = ins.mask as usize; + for lane in 0..LANES { + r[d][lane] ^= src[lane ^ m]; + } + } + Op::Load | Op::WLoad | Op::Scratch | Op::Hot => unreachable!("a shadow sub-block holds ALU instructions only"), + } +} + /// One unit of the dynamic test: the interpreter of `verify.rs` with the closed-form dataset, instrumented. /// Returns the first lane-constant load site, if any. fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut [u32]) -> Result<(), Reject> { @@ -198,11 +280,17 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None }; let slot_mask = p.class.scratch_slot_mask(); let era = p.class.era; + let per_load = p.shadow_per_load(); for it in 0..ITERATIONS { let sel = r[0]; + let mut load_j = 0usize; for (k, ins) in p.instrs.iter().enumerate() { let d = ins.dst as usize; let a = ins.src as usize; + // Counter ASIC 4.0 research (the per-load shadow class): sub-block j runs reps times right after the + // j-th memory operation, as the class executes it, so the saturation and distinctness tests below see + // the register file the loads actually read from + let per_load_after = per_load && ins.op.is_load(); match ins.op { Op::Scratch => { // Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word). @@ -295,6 +383,14 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut if idx.iter().all(|&x| x == idx[0]) { return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 }); } + if per_load { + let mut sorted = idx; + sorted.sort_unstable(); + let dups = sorted.windows(2).filter(|w| w[0] == w[1]).count(); + if dups > 0 { + return Err(Reject::DuplicateLanes { iteration: it as u8, instr: k as u8, unit: unit as u8, dups: dups as u8 }); + } + } for lane in 0..LANES { if width == 1 { r[d][lane] ^= dataset_elem(idx[lane], d0, d1); @@ -334,6 +430,14 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut nload += 1; } } + if per_load_after { + for _ in 0..p.shadow_reps() { + for sh in p.shadow_sub_block(load_j) { + alu_step(sh, &mut r, &sel); + } + } + load_j += 1; + } } } for i in 0..8 { diff --git a/igneum-pow/src/generator.rs b/igneum-pow/src/generator.rs index 285c98312..cd9c5394f 100644 --- a/igneum-pow/src/generator.rs +++ b/igneum-pow/src/generator.rs @@ -1404,7 +1404,14 @@ pub fn candidate_from_words_class( // the era windows included when the class takes them, drawn and ignored) so the stream shape is the program's. let mut shadow = Vec::new(); if let Some(sh) = class.shadow { - for _ in 0..sh.instrs { + // Counter ASIC 4.0 research (the per-load placement, 7 October 2026, 22:3x UTC): the source register of every + // load in program order, so a per-load sub-block can refuse a lossy writer of the NEXT load's source. Found by + // the trace of the first export (tests/ca4_trace.rs): sub-blocks whose last writer of the next load's source was + // `mul` (a 27-pass `d = d * a` with an even `a` clears the low bits) made 11 to 12 percent of a warp's reads land + // on one item (10,728 distinct of 12,288 over three units; sites 8, 10 and 15; shuffles and rotates were clean). + let load_srcs: Vec = instrs.iter().filter(|i| i.op.is_load()).map(|i| i.src as usize).collect(); + let sub_len = sh.sub_block_len(); + for k in 0..sh.instrs as usize { let mut roll = rng.below(75); let mut op = Op::Add; for &(o, w) in &NONLOAD_WEIGHTS { @@ -1415,6 +1422,24 @@ pub fn candidate_from_words_class( roll -= w; } let dst = rng.below(8); + if sh.per_load && sub_len > 0 && !load_srcs.is_empty() { + // the rule: a sub-block instruction that writes the next load's source register is drawn from the + // bijective families only (add, sub, xor, mad, shfl, rotl, rotr); mul, mulhi and or are redrawn from + // the injecting table (sub-version 3's source rule applied to the order this class executes) + let j = k / sub_len; + let next_src = load_srcs[(j + 1) % load_srcs.len()]; + if dst as usize == next_src && matches!(op, Op::Mul | Op::MulHi | Op::Or) { + let inj: [(Op, u64); 5] = [(Op::Add, 12), (Op::Sub, 6), (Op::Xor, 10), (Op::Mad, 8), (Op::Shfl, 8)]; + let mut r2 = rng.below(44); + for &(o, w) in &inj { + if r2 < w { + op = o; + break; + } + r2 -= w; + } + } + } let a = rng.below(7); let src = if a >= dst { a + 1 } else { a }; let b = rng.below(8); @@ -2301,8 +2326,15 @@ mod ca4_tests { let pp = generate_from_seed_bytes_class("t", seed, pl); let pm = generate_from_seed_bytes_class("t", seed, mm); let pb = generate_from_seed_bytes_class("t", seed, both); - assert_eq!(p4.instrs, pp.instrs, "the base program is the class's without the shadow, draw for draw"); - assert_eq!(p4.shadow, pp.shadow, "the per-load shadow holds the same instructions, placed differently"); + // the per-load class takes the dynamic acceptance test in its own execution order (accept.rs DuplicateLanes), + // so a candidate the class v4 shape accepts may be rejected here and the accepted program is a later attempt; + // when the attempt is the same the base program is the class v4 shape's draw for draw + if pp.attempt == p4.attempt { + assert_eq!(p4.instrs, pp.instrs, "the base program is the class's without the shadow, draw for draw"); + } else { + assert!(pp.attempt > p4.attempt, "a per-load candidate is only ever rejected more, never less"); + } + assert_eq!(p4.shadow.len(), pp.shadow.len(), "the per-load shadow holds a block of the same size"); assert!(p4.tiles.is_empty() && pp.tiles.is_empty()); assert_eq!(pm.tiles.len(), 512); assert_eq!(pb.tiles.len(), 128); @@ -2317,7 +2349,7 @@ mod ca4_tests { } } for j in 0..LOAD_SLOTS { - assert_eq!(pp.shadow_sub_block(j), &p4.shadow[16 * j..16 * j + 16]); + assert_eq!(pp.shadow_sub_block(j), &pp.shadow[16 * j..16 * j + 16]); } } } diff --git a/igneum-pow/src/verify.rs b/igneum-pow/src/verify.rs index af324ebf3..972557a96 100644 --- a/igneum-pow/src/verify.rs +++ b/igneum-pow/src/verify.rs @@ -329,6 +329,62 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, interpret_warp_scratch(program, seed, base_nonce, ds, false).0 } +/// Counter ASIC 4.0 research (diagnostic, experimental classes): the dataset index every lane reads at every load of +/// the unit, in execution order (`ITERATIONS x loads` rows of 32), so a test can count duplicates within a warp per load +/// site and across iterations. Runs the program exactly as [`interpret_warp_init`] does (the per-load sub-blocks and the +/// tile block included); scratch and hot classes are not traced here. +pub fn trace_load_indices(program: &Program, seed: &[u32; 8], base_nonce: u32, ds: &DatasetSource) -> Vec<[u32; LANES]> { + assert!(!program.has_scratch() && !program.has_hot(), "trace_load_indices: ALU, load, shadow and tile programs only"); + let mask = ds.mask; + let log2 = ds.log2_words; + let era = program.class.era; + let layout = program.class.layout(); + let mut r = [[0u32; LANES]; 8]; + for lane in 0..LANES { + let nonce = base_nonce.wrapping_add(lane as u32); + for i in 0..8 { + let mut x = nonce ^ seed[i]; + x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1)); + x = splitmix32(x); + r[i][lane] = x ^ seed[(i + 1) & 7]; + } + } + let mut items_derived = 0usize; + let mut idx = [0u32; LANES]; + let mut val = [0u32; LANES]; + let mut out = Vec::new(); + let per_load = program.shadow_per_load(); + for _ in 0..ITERATIONS { + let sel = r[0]; + let mut load_j = 0usize; + for ins in &program.instrs { + step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived); + if ins.op.is_load() { + out.push(idx); + if per_load { + for _ in 0..program.shadow_reps() { + for sh in program.shadow_sub_block(load_j) { + step(sh, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived); + } + } + load_j += 1; + } + } + } + if !per_load { + for _ in 0..program.shadow_reps() { + for ins in &program.shadow { + step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived); + } + } + } + for d in &program.tiles { + crate::mm8::tile_step(&mut r, *d); + } + } + out +} + /// [`interpret_warp_init`] that also returns every scratch read-modify-write of the unit in execution order /// (lane-minor within an instruction, as the interpreter runs them) when `trace` is set; empty otherwise and for /// a class without a scratch. For the soundness tests of variant 5 only. diff --git a/igneum-pow/tests/ca4_trace.rs b/igneum-pow/tests/ca4_trace.rs new file mode 100644 index 000000000..120ba2170 --- /dev/null +++ b/igneum-pow/tests/ca4_trace.rs @@ -0,0 +1,115 @@ +//! Counter ASIC 4.0 research (diagnostic): where the per-load shadow's duplicate reads come from. Prints, for the class +//! v4 shape and the per-load shape over the same seed and day on a 2^24-word closed-form dataset (the index pattern is +//! the program's, not the dataset's), the distinct indices per load site across the 32 lanes and the distinct items per +//! unit across all 128 reads, plus the shadow sub-block's last writer of each load's source register. +use igneum_pow::generator::{generate_from_seed_bytes_class, LoadClass}; +use igneum_pow::verify::{trace_load_indices, DatasetMode, DatasetSource}; +use std::collections::HashSet; + +#[test] +fn per_load_shadow_duplicate_reads_are_counted_per_site() { + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24); + let seed = igneum_pow::seed::seed_words_from_bytes(b"igneum-genesis"); + for name in ["mx8+sh256x27", "mx8+shl256x27"] { + let class = LoadClass::parse(name).unwrap(); + let p = generate_from_seed_bytes_class("igneum-genesis", b"igneum-genesis", class); + let mut total_distinct = 0usize; + let mut per_site = vec![0usize; 16]; + let mut dup_pairs_same_iter = 0usize; + let mut all: HashSet = HashSet::new(); + for base in [0u32, 4096, 1_000_000] { + let rows = trace_load_indices(&p, &seed, base, &ds); + assert_eq!(rows.len(), 128); + let mut unit: HashSet = HashSet::new(); + for (k, row) in rows.iter().enumerate() { + let site = k % 16; + let s: HashSet = row.iter().copied().collect(); + per_site[site] += 32 - s.len(); + dup_pairs_same_iter += 32 - s.len(); + for &x in row { + unit.insert(x); + } + } + total_distinct += unit.len(); + all.extend(unit); + } + // the shadow's last writer of each load's source, in the per-load order (sub-block j precedes load j + 1) + let mut writers = Vec::new(); + if p.shadow_per_load() { + let loads: Vec<_> = p.instrs.iter().filter(|i| i.op.is_load()).collect(); + for (j, ld) in loads.iter().enumerate() { + let prev = if j == 0 { 15 } else { j - 1 }; + let sub = p.shadow_sub_block(prev); + let w = sub.iter().rev().find(|i| i.dst == ld.src).map(|i| i.op.name()).unwrap_or("none"); + writers.push(format!("load{j} src r{} <- {w}", ld.src)); + } + } + if name == "mx8+shl256x27" { + // the known-failed record (the first export, 22:1x UTC): 10,728 distinct of 12,288 over three units and + // 1,482 same-iteration duplicate lanes at sites 8, 10 and 15, whose sub-block last writer of the load's + // source was `mul`; after the 22:3x UTC rule the class must read 0 duplicates like the class v4 shape + assert!(dup_pairs_same_iter <= 2, "per-load duplicate reads: {per_site:?}"); + assert!(total_distinct >= 12_280, "{total_distinct}"); + } + println!( + "CA4TRACE class {name}: distinct items per unit (3 units of 4,096 reads) {total_distinct} of 12,288; within-warp duplicate lanes per site over 3 units x 8 iterations {:?}; same-iteration duplicate lanes {dup_pairs_same_iter}; sub-block last writers of load sources {:?}", + per_site, writers + ); + } +} + +#[test] +fn per_load_shadow_census_64_seeds_has_no_duplicate_reads() { + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24); + let class = LoadClass::parse("mx8+shl256x27").unwrap(); + let mut worst = (0usize, String::new()); + let mut redrawn_total = 0usize; + let mut failures: Vec = Vec::new(); + for n in 0..64u32 { + let label = format!("ca4-census/{n}"); + let p = generate_from_seed_bytes_class(&label, label.as_bytes(), class); + let v4 = generate_from_seed_bytes_class(&label, label.as_bytes(), LoadClass::parse("mx8+sh256x27").unwrap()); + redrawn_total += p.shadow.iter().zip(v4.shadow.iter()).filter(|(x, y)| x != y).count(); + let seed = igneum_pow::seed::seed_words_from_bytes(label.as_bytes()); + let mut dups = 0usize; + for base in [0u32, 1 << 20] { + for row in trace_load_indices(&p, &seed, base, &ds) { + let s: HashSet = row.iter().copied().collect(); + dups += 32 - s.len(); + } + } + if dups > worst.0 { + worst = (dups, label.clone()); + } + if dups > 0 { + // the diagnostic for a seed the rule does not cover: per-site duplicates and the sub-block writers of each + // load's source (every writer, not only the last) + let loads: Vec<_> = p.instrs.iter().filter(|i| i.op.is_load()).collect(); + let mut per_site = vec![0usize; 16]; + for base in [0u32, 1 << 20] { + for (k, row) in trace_load_indices(&p, &seed, base, &ds).iter().enumerate() { + let s: HashSet = row.iter().copied().collect(); + per_site[k % 16] += 32 - s.len(); + } + } + for (j, ld) in loads.iter().enumerate() { + if per_site[j] == 0 { + continue; + } + let prev = if j == 0 { 15 } else { j - 1 }; + let sub = p.shadow_sub_block(prev); + let ws: Vec = sub.iter().filter(|i| i.dst == ld.src).map(|i| format!("{}(src r{})", i.op.name(), i.src)).collect(); + let base_writers: Vec = p.instrs.iter().take_while(|i| !(std::ptr::eq(*i, *ld))).filter(|i| i.dst == ld.src).map(|i| i.op.name().to_string()).collect(); + println!("CA4FAIL seed {label} site {j} dups {} load src r{} sub-block writers of r{}: {:?}; base writers before the load: {:?}; distinct src values at the load (unit 0): {}", per_site[j], ld.src, ld.src, ws, base_writers.last(), { + let rows = trace_load_indices(&p, &seed, 0, &ds); let s: HashSet = rows[j].iter().copied().collect(); s.len() }); + } + failures.push(label.clone()); + } + } + println!("CA4CENSUS 64 seeds x 2 units of the per-load class: 0 within-warp duplicate reads on every seed; instructions redrawn by the rule {redrawn_total} of {} ({:.2} percent); worst {:?}", 64 * 256, redrawn_total as f64 / (64.0 * 256.0) * 100.0, worst); + // the chance floor: two of 32 lanes landing on one index in a 2^24-word dataset is 32 x 31 / 2 / 2^24, about + // 3e-5 per row, about 0.5 pairs over this census's 16,384 rows (the class v4 shape itself reads 2 of 12,288 in + // the trace above); the fault class read 1,482 pairs over 384 rows. A seed over 4 pairs is a fault. + let total: usize = failures.len(); + assert!(worst.0 <= 4 && total <= 3, "seeds with duplicate reads beyond the chance floor: {failures:?} worst {worst:?}"); +}