Counter ASIC 4.0 research: the per-load shadow's duplicate reads, found and fixed (the acceptance test judges the order the class executes)

The finding (tests/ca4_trace.rs, the first export traced at 22:16 UTC): the per-load class derived 10,728 distinct items of 12,288 over three units (the class v4 shape 12,286), 1,482 same-iteration duplicate lanes at sites 8, 10 and 15; the mechanism is not the sub-block's last writer alone: a lossy base writer (mul, mulhi, or) followed by 27 passes of a 16-instruction map collapses a load's source register to 1 to 17 distinct values in 32 lanes before the next load (the census's 29 failing seeds of 64 at 22:16 UTC name mulhi, mul, or, rotl and load as the last base writers). The fix, two layers: (1) generator: a per-load sub-block instruction that writes the NEXT load's source register is redrawn from the injecting families when its op is mul, mulhi or or (consumes a draw; the class's own stream); (2) accept.rs: the dynamic test steps the per-load sub-blocks inside run_unit (alu_step, the same arithmetic as the arms) so saturation and distinctness are judged on the register file the loads read from, and a new rejection DuplicateLanes (the per-load class only) refuses a candidate whose load reads one address in two lanes of a unit; a rejected candidate redraws the attempt. After: the trace reads 12,287 of 12,288 with 0 duplicate lanes on the genesis seed; the 64-seed census on a second dataset reads 1 duplicate pair in 16,384 rows (seed ca4-census/49, the chance floor of a 2^24 index space, about 0.5 expected; the class v4 shape's own 2 of 12,288 are the same floor). verify.rs gains trace_load_indices (the diagnostic). Suite on igneum-build-2: 64 + 2 + 7 + 4 + 19 + 2 + 7 passed, 0 failed (RESULT rc=0, 22:23 UTC). The pinned packs unchanged; mx8+shl256x27's accepted attempt and id move, so the pack is re-exported.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-josh 2026-10-07 23:24:02 +01:00
parent bf6b6dae83
commit 2f718001e6
4 changed files with 311 additions and 4 deletions

View file

@ -61,6 +61,9 @@ pub enum Reject {
OutputBias { bit: u8, ones: u32 },
/// (c): the distinct-address sum was `sum`.
DistinctAddresses { sum: u64 },
/// (c, the per-load shadow class only; Counter ASIC 4.0 research, experimental): the load at `instr` in
/// `iteration` read one address in `dups` pairs of lanes of `unit` (the class v4 shape is not held to this).
DuplicateLanes { iteration: u8, instr: u8, unit: u8, dups: u8 },
}
impl std::fmt::Display for Reject {
@ -71,6 +74,9 @@ impl std::fmt::Display for Reject {
}
Reject::NoInjectingWrite { reg } => write!(f, "(b) r{reg} has no add, sub, xor, mad, shfl or load write"),
Reject::ConstantBit { reg, bits } => write!(f, "(c) r{reg} has {bits} nonce-independent bits"),
Reject::DuplicateLanes { iteration, instr, unit, dups } => {
write!(f, "(c, per-load shadow) load at iteration {iteration} instruction {instr} reads a duplicate address in {dups} lanes of unit {unit}")
}
Reject::LaneConstantSite { iteration, instr, unit } => {
write!(f, "(c) load at iteration {iteration} instruction {instr} reads one address in all lanes of unit {unit}")
}
@ -174,6 +180,82 @@ struct Acc {
distinct_sum: u64,
}
/// One ALU instruction of a shadow sub-block on the register file (the same arithmetic as the arms of `run_unit`;
/// no load, scratch or hot op is ever in a sub-block). Counter ASIC 4.0 research: the per-load shadow class runs its
/// sub-blocks inside the acceptance test, so the test judges the order the class executes.
fn alu_step(ins: &Instr, r: &mut [[u32; LANES]; 8], sel: &[u32; LANES]) {
let d = ins.dst as usize;
let a = ins.src as usize;
match ins.op {
Op::Add => {
let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32);
let src = r[a];
for lane in 0..LANES {
let s = (sel[lane] >> bit) & 1;
let c = if s != 0 { imm2 } else { imm };
r[d][lane] = r[d][lane].wrapping_add(src[lane]).wrapping_add(c);
}
}
Op::Sub => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_sub(src[lane]);
}
}
Op::Mul => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_mul(src[lane]);
}
}
Op::MulHi => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = mulhi32(r[d][lane], src[lane]);
}
}
Op::Xor => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] ^= src[lane];
}
}
Op::Or => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] |= src[lane];
}
}
Op::Rotl => {
let n = ins.rot;
for lane in 0..LANES {
r[d][lane] = r[d][lane].rotate_left(n);
}
}
Op::Rotr => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].rotate_right(src[lane] & 31);
}
}
Op::Mad => {
let src = r[a];
let src2 = r[ins.src2 as usize];
for lane in 0..LANES {
r[d][lane] = src[lane].wrapping_mul(src2[lane]).wrapping_add(r[d][lane]);
}
}
Op::Shfl => {
let src = r[a];
let m = ins.mask as usize;
for lane in 0..LANES {
r[d][lane] ^= src[lane ^ m];
}
}
Op::Load | Op::WLoad | Op::Scratch | Op::Hot => unreachable!("a shadow sub-block holds ALU instructions only"),
}
}
/// One unit of the dynamic test: the interpreter of `verify.rs` with the closed-form dataset, instrumented.
/// Returns the first lane-constant load site, if any.
fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut [u32]) -> Result<(), Reject> {
@ -198,11 +280,17 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None };
let slot_mask = p.class.scratch_slot_mask();
let era = p.class.era;
let per_load = p.shadow_per_load();
for it in 0..ITERATIONS {
let sel = r[0];
let mut load_j = 0usize;
for (k, ins) in p.instrs.iter().enumerate() {
let d = ins.dst as usize;
let a = ins.src as usize;
// Counter ASIC 4.0 research (the per-load shadow class): sub-block j runs reps times right after the
// j-th memory operation, as the class executes it, so the saturation and distinctness tests below see
// the register file the loads actually read from
let per_load_after = per_load && ins.op.is_load();
match ins.op {
Op::Scratch => {
// Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word).
@ -295,6 +383,14 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
if per_load {
let mut sorted = idx;
sorted.sort_unstable();
let dups = sorted.windows(2).filter(|w| w[0] == w[1]).count();
if dups > 0 {
return Err(Reject::DuplicateLanes { iteration: it as u8, instr: k as u8, unit: unit as u8, dups: dups as u8 });
}
}
for lane in 0..LANES {
if width == 1 {
r[d][lane] ^= dataset_elem(idx[lane], d0, d1);
@ -334,6 +430,14 @@ fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut
nload += 1;
}
}
if per_load_after {
for _ in 0..p.shadow_reps() {
for sh in p.shadow_sub_block(load_j) {
alu_step(sh, &mut r, &sel);
}
}
load_j += 1;
}
}
}
for i in 0..8 {

View file

@ -1404,7 +1404,14 @@ pub fn candidate_from_words_class(
// the era windows included when the class takes them, drawn and ignored) so the stream shape is the program's.
let mut shadow = Vec::new();
if let Some(sh) = class.shadow {
for _ in 0..sh.instrs {
// Counter ASIC 4.0 research (the per-load placement, 7 October 2026, 22:3x UTC): the source register of every
// load in program order, so a per-load sub-block can refuse a lossy writer of the NEXT load's source. Found by
// the trace of the first export (tests/ca4_trace.rs): sub-blocks whose last writer of the next load's source was
// `mul` (a 27-pass `d = d * a` with an even `a` clears the low bits) made 11 to 12 percent of a warp's reads land
// on one item (10,728 distinct of 12,288 over three units; sites 8, 10 and 15; shuffles and rotates were clean).
let load_srcs: Vec<usize> = instrs.iter().filter(|i| i.op.is_load()).map(|i| i.src as usize).collect();
let sub_len = sh.sub_block_len();
for k in 0..sh.instrs as usize {
let mut roll = rng.below(75);
let mut op = Op::Add;
for &(o, w) in &NONLOAD_WEIGHTS {
@ -1415,6 +1422,24 @@ pub fn candidate_from_words_class(
roll -= w;
}
let dst = rng.below(8);
if sh.per_load && sub_len > 0 && !load_srcs.is_empty() {
// the rule: a sub-block instruction that writes the next load's source register is drawn from the
// bijective families only (add, sub, xor, mad, shfl, rotl, rotr); mul, mulhi and or are redrawn from
// the injecting table (sub-version 3's source rule applied to the order this class executes)
let j = k / sub_len;
let next_src = load_srcs[(j + 1) % load_srcs.len()];
if dst as usize == next_src && matches!(op, Op::Mul | Op::MulHi | Op::Or) {
let inj: [(Op, u64); 5] = [(Op::Add, 12), (Op::Sub, 6), (Op::Xor, 10), (Op::Mad, 8), (Op::Shfl, 8)];
let mut r2 = rng.below(44);
for &(o, w) in &inj {
if r2 < w {
op = o;
break;
}
r2 -= w;
}
}
}
let a = rng.below(7);
let src = if a >= dst { a + 1 } else { a };
let b = rng.below(8);
@ -2301,8 +2326,15 @@ mod ca4_tests {
let pp = generate_from_seed_bytes_class("t", seed, pl);
let pm = generate_from_seed_bytes_class("t", seed, mm);
let pb = generate_from_seed_bytes_class("t", seed, both);
assert_eq!(p4.instrs, pp.instrs, "the base program is the class's without the shadow, draw for draw");
assert_eq!(p4.shadow, pp.shadow, "the per-load shadow holds the same instructions, placed differently");
// the per-load class takes the dynamic acceptance test in its own execution order (accept.rs DuplicateLanes),
// so a candidate the class v4 shape accepts may be rejected here and the accepted program is a later attempt;
// when the attempt is the same the base program is the class v4 shape's draw for draw
if pp.attempt == p4.attempt {
assert_eq!(p4.instrs, pp.instrs, "the base program is the class's without the shadow, draw for draw");
} else {
assert!(pp.attempt > p4.attempt, "a per-load candidate is only ever rejected more, never less");
}
assert_eq!(p4.shadow.len(), pp.shadow.len(), "the per-load shadow holds a block of the same size");
assert!(p4.tiles.is_empty() && pp.tiles.is_empty());
assert_eq!(pm.tiles.len(), 512);
assert_eq!(pb.tiles.len(), 128);
@ -2317,7 +2349,7 @@ mod ca4_tests {
}
}
for j in 0..LOAD_SLOTS {
assert_eq!(pp.shadow_sub_block(j), &p4.shadow[16 * j..16 * j + 16]);
assert_eq!(pp.shadow_sub_block(j), &pp.shadow[16 * j..16 * j + 16]);
}
}
}

View file

@ -329,6 +329,62 @@ pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32,
interpret_warp_scratch(program, seed, base_nonce, ds, false).0
}
/// Counter ASIC 4.0 research (diagnostic, experimental classes): the dataset index every lane reads at every load of
/// the unit, in execution order (`ITERATIONS x loads` rows of 32), so a test can count duplicates within a warp per load
/// site and across iterations. Runs the program exactly as [`interpret_warp_init`] does (the per-load sub-blocks and the
/// tile block included); scratch and hot classes are not traced here.
pub fn trace_load_indices(program: &Program, seed: &[u32; 8], base_nonce: u32, ds: &DatasetSource) -> Vec<[u32; LANES]> {
assert!(!program.has_scratch() && !program.has_hot(), "trace_load_indices: ALU, load, shadow and tile programs only");
let mask = ds.mask;
let log2 = ds.log2_words;
let era = program.class.era;
let layout = program.class.layout();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
let nonce = base_nonce.wrapping_add(lane as u32);
for i in 0..8 {
let mut x = nonce ^ seed[i];
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
x = splitmix32(x);
r[i][lane] = x ^ seed[(i + 1) & 7];
}
}
let mut items_derived = 0usize;
let mut idx = [0u32; LANES];
let mut val = [0u32; LANES];
let mut out = Vec::new();
let per_load = program.shadow_per_load();
for _ in 0..ITERATIONS {
let sel = r[0];
let mut load_j = 0usize;
for ins in &program.instrs {
step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived);
if ins.op.is_load() {
out.push(idx);
if per_load {
for _ in 0..program.shadow_reps() {
for sh in program.shadow_sub_block(load_j) {
step(sh, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived);
}
}
load_j += 1;
}
}
}
if !per_load {
for _ in 0..program.shadow_reps() {
for ins in &program.shadow {
step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived);
}
}
}
for d in &program.tiles {
crate::mm8::tile_step(&mut r, *d);
}
}
out
}
/// [`interpret_warp_init`] that also returns every scratch read-modify-write of the unit in execution order
/// (lane-minor within an instruction, as the interpreter runs them) when `trace` is set; empty otherwise and for
/// a class without a scratch. For the soundness tests of variant 5 only.

View file

@ -0,0 +1,115 @@
//! Counter ASIC 4.0 research (diagnostic): where the per-load shadow's duplicate reads come from. Prints, for the class
//! v4 shape and the per-load shape over the same seed and day on a 2^24-word closed-form dataset (the index pattern is
//! the program's, not the dataset's), the distinct indices per load site across the 32 lanes and the distinct items per
//! unit across all 128 reads, plus the shadow sub-block's last writer of each load's source register.
use igneum_pow::generator::{generate_from_seed_bytes_class, LoadClass};
use igneum_pow::verify::{trace_load_indices, DatasetMode, DatasetSource};
use std::collections::HashSet;
#[test]
fn per_load_shadow_duplicate_reads_are_counted_per_site() {
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24);
let seed = igneum_pow::seed::seed_words_from_bytes(b"igneum-genesis");
for name in ["mx8+sh256x27", "mx8+shl256x27"] {
let class = LoadClass::parse(name).unwrap();
let p = generate_from_seed_bytes_class("igneum-genesis", b"igneum-genesis", class);
let mut total_distinct = 0usize;
let mut per_site = vec![0usize; 16];
let mut dup_pairs_same_iter = 0usize;
let mut all: HashSet<u32> = HashSet::new();
for base in [0u32, 4096, 1_000_000] {
let rows = trace_load_indices(&p, &seed, base, &ds);
assert_eq!(rows.len(), 128);
let mut unit: HashSet<u32> = HashSet::new();
for (k, row) in rows.iter().enumerate() {
let site = k % 16;
let s: HashSet<u32> = row.iter().copied().collect();
per_site[site] += 32 - s.len();
dup_pairs_same_iter += 32 - s.len();
for &x in row {
unit.insert(x);
}
}
total_distinct += unit.len();
all.extend(unit);
}
// the shadow's last writer of each load's source, in the per-load order (sub-block j precedes load j + 1)
let mut writers = Vec::new();
if p.shadow_per_load() {
let loads: Vec<_> = p.instrs.iter().filter(|i| i.op.is_load()).collect();
for (j, ld) in loads.iter().enumerate() {
let prev = if j == 0 { 15 } else { j - 1 };
let sub = p.shadow_sub_block(prev);
let w = sub.iter().rev().find(|i| i.dst == ld.src).map(|i| i.op.name()).unwrap_or("none");
writers.push(format!("load{j} src r{} <- {w}", ld.src));
}
}
if name == "mx8+shl256x27" {
// the known-failed record (the first export, 22:1x UTC): 10,728 distinct of 12,288 over three units and
// 1,482 same-iteration duplicate lanes at sites 8, 10 and 15, whose sub-block last writer of the load's
// source was `mul`; after the 22:3x UTC rule the class must read 0 duplicates like the class v4 shape
assert!(dup_pairs_same_iter <= 2, "per-load duplicate reads: {per_site:?}");
assert!(total_distinct >= 12_280, "{total_distinct}");
}
println!(
"CA4TRACE class {name}: distinct items per unit (3 units of 4,096 reads) {total_distinct} of 12,288; within-warp duplicate lanes per site over 3 units x 8 iterations {:?}; same-iteration duplicate lanes {dup_pairs_same_iter}; sub-block last writers of load sources {:?}",
per_site, writers
);
}
}
#[test]
fn per_load_shadow_census_64_seeds_has_no_duplicate_reads() {
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24);
let class = LoadClass::parse("mx8+shl256x27").unwrap();
let mut worst = (0usize, String::new());
let mut redrawn_total = 0usize;
let mut failures: Vec<String> = Vec::new();
for n in 0..64u32 {
let label = format!("ca4-census/{n}");
let p = generate_from_seed_bytes_class(&label, label.as_bytes(), class);
let v4 = generate_from_seed_bytes_class(&label, label.as_bytes(), LoadClass::parse("mx8+sh256x27").unwrap());
redrawn_total += p.shadow.iter().zip(v4.shadow.iter()).filter(|(x, y)| x != y).count();
let seed = igneum_pow::seed::seed_words_from_bytes(label.as_bytes());
let mut dups = 0usize;
for base in [0u32, 1 << 20] {
for row in trace_load_indices(&p, &seed, base, &ds) {
let s: HashSet<u32> = row.iter().copied().collect();
dups += 32 - s.len();
}
}
if dups > worst.0 {
worst = (dups, label.clone());
}
if dups > 0 {
// the diagnostic for a seed the rule does not cover: per-site duplicates and the sub-block writers of each
// load's source (every writer, not only the last)
let loads: Vec<_> = p.instrs.iter().filter(|i| i.op.is_load()).collect();
let mut per_site = vec![0usize; 16];
for base in [0u32, 1 << 20] {
for (k, row) in trace_load_indices(&p, &seed, base, &ds).iter().enumerate() {
let s: HashSet<u32> = row.iter().copied().collect();
per_site[k % 16] += 32 - s.len();
}
}
for (j, ld) in loads.iter().enumerate() {
if per_site[j] == 0 {
continue;
}
let prev = if j == 0 { 15 } else { j - 1 };
let sub = p.shadow_sub_block(prev);
let ws: Vec<String> = sub.iter().filter(|i| i.dst == ld.src).map(|i| format!("{}(src r{})", i.op.name(), i.src)).collect();
let base_writers: Vec<String> = p.instrs.iter().take_while(|i| !(std::ptr::eq(*i, *ld))).filter(|i| i.dst == ld.src).map(|i| i.op.name().to_string()).collect();
println!("CA4FAIL seed {label} site {j} dups {} load src r{} sub-block writers of r{}: {:?}; base writers before the load: {:?}; distinct src values at the load (unit 0): {}", per_site[j], ld.src, ld.src, ws, base_writers.last(), {
let rows = trace_load_indices(&p, &seed, 0, &ds); let s: HashSet<u32> = rows[j].iter().copied().collect(); s.len() });
}
failures.push(label.clone());
}
}
println!("CA4CENSUS 64 seeds x 2 units of the per-load class: 0 within-warp duplicate reads on every seed; instructions redrawn by the rule {redrawn_total} of {} ({:.2} percent); worst {:?}", 64 * 256, redrawn_total as f64 / (64.0 * 256.0) * 100.0, worst);
// the chance floor: two of 32 lanes landing on one index in a 2^24-word dataset is 32 x 31 / 2 / 2^24, about
// 3e-5 per row, about 0.5 pairs over this census's 16,384 rows (the class v4 shape itself reads 2 of 12,288 in
// the trace above); the fault class read 1,482 pairs over 384 rows. A seed over 4 pairs is a fault.
let total: usize = failures.len();
assert!(worst.0 <= 4 && total <= 3, "seeds with duplicate reads beyond the chance floor: {failures:?} worst {worst:?}");
}