A prototype behind a new LoadClass field (derive_len) and Shape field, Shape::for_class_day: every mixer slot of the item derivation runs a straight-line program of 736 instructions drawn from the day key stream (the same SplitMix64 stream, after the 40 mixer draws), twelve two-register forms, the chain rule of SuperscalarHash made strict (every instruction reads the register the previous one wrote), an acceptance test with the x8 mixer's operation and multiply counts from the code as floors (72 x 144 as written, 72 x 128 hoisted, 1,152 multiplies). The verifier runs the program with a word-major (SoA) interpreter over the 32 items of a load, dispatching on instruction pairs; no JIT. The emitter writes mh_round_0..8 into memhard.h, memhard.metal and kernel.cl. Packs dr736-genesis and dr736-devnet-epoch0 under proto-cuda/packs-ca3-derive. The v2 and v3 paths are untouched: every pinned pack re-exports byte for byte (tests/packs.rs), cargo test -p igneum-pow 58 + 7 + 4 + 19 + 7 green. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
51 lines
2.6 KiB
Rust
51 lines
2.6 KiB
Rust
//! The derivation interpreter's cost per instruction per batch (Counter ASIC 3.0 item 2): the day's program against
|
|
//! a uniform program of the same length (every instruction one form, so the dispatch is predictable), which splits
|
|
//! the per-instruction cost into the dispatch and the vector body. A functional tool, not a bench-log number on its
|
|
//! own: run it under the measure lock and state the load average when a figure is recorded.
|
|
//! cargo run --release --example derive_perf [len] [reps]
|
|
use igneum_pow::derive::{run_round, DInstr, DOp, DeriveProgram, SoaState, DERIVE_LEN_X8, DERIVE_REGS, SOA_LANES};
|
|
use igneum_pow::seed::SplitMix64;
|
|
use std::time::Instant;
|
|
|
|
fn time(prog: &DeriveProgram, reps: usize) -> (f64, u64) {
|
|
let mut st: SoaState = [[0u32; SOA_LANES]; DERIVE_REGS];
|
|
let mut x = SplitMix64::new(5);
|
|
for r in 0..DERIVE_REGS {
|
|
for k in 0..SOA_LANES {
|
|
st[r][k] = x.next() as u32;
|
|
}
|
|
}
|
|
let t = Instant::now();
|
|
for _ in 0..reps {
|
|
for p in &prog.rounds {
|
|
run_round(p, &mut st);
|
|
}
|
|
}
|
|
let ns = t.elapsed().as_nanos() as f64 / (reps as f64 * prog.instr_count() as f64);
|
|
(ns, st[0][0] as u64 ^ st[15][31] as u64)
|
|
}
|
|
|
|
fn uniform(len: u32, op: DOp) -> DeriveProgram {
|
|
let mut p = DeriveProgram::draw_candidate(&mut SplitMix64::new(1), len, 0);
|
|
for ins in p.rounds.iter_mut().flatten() {
|
|
*ins = DInstr { op, rot: if op.has_rot() { 13 } else { 0 }, imm: if op.has_imm() { 0x9e37_79b9 } else { 0 }, ..*ins };
|
|
}
|
|
p
|
|
}
|
|
|
|
fn main() {
|
|
let a: Vec<String> = std::env::args().collect();
|
|
let len: u32 = a.get(1).and_then(|s| s.parse().ok()).unwrap_or(DERIVE_LEN_X8);
|
|
let reps: usize = a.get(2).and_then(|s| s.parse().ok()).unwrap_or(400);
|
|
let real = DeriveProgram::draw(&mut SplitMix64::new(0x3067619f3c269176), len);
|
|
let per_item = real.instr_count();
|
|
println!("len {len}: {per_item} instructions per item, {} GPU ops, {} chip ops, {} multiplies; batch of {SOA_LANES} lanes, {reps} reps", real.gpu_ops(), real.chip_ops(), real.muls());
|
|
for round in 0..2 {
|
|
let (ns, sink) = time(&real, reps);
|
|
println!("round {round}: drawn program {ns:.2} ns per instruction per batch ({:.1} us per batch, {:.2} ms per 128 batches) sink {sink:x}", ns * per_item as f64 / 1e3, ns * per_item as f64 * 128.0 / 1e6);
|
|
for op in [DOp::Add, DOp::Mul, DOp::XRot, DOp::MulC, DOp::AndX] {
|
|
let (ns, sink) = time(&uniform(len, op), reps);
|
|
println!("round {round}: uniform {:5} {ns:.2} ns per instruction per batch sink {sink:x}", op.name());
|
|
}
|
|
}
|
|
}
|