class-v6: V6-02 re-done: the executed load and byte counters (loads_per_hash_executed, bytes_per_hash_executed) are read by the emitters alone; the acceptance and the draw read the drawn count as before (the first form, c462b526e, routed the executed count into the acceptance and moved every reg64 verdict: hl-v6-all drew 0x1626c5853aa84261 instead of 0x4de7b836cc40a4ea, found by the re-export's read-back)
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
parent
d47d66a58e
commit
efe06c9dbc
2 changed files with 28 additions and 18 deletions
|
|
@ -355,8 +355,8 @@ fn class_header_lines(p: &Program) -> String {
|
|||
let c = p.width_counts();
|
||||
s.push_str(&format!("#define IGNEUM_LOAD_WIDTH_COUNTS {{ {}, {}, {} }} // loads of 4, 16, 64 bytes per program
|
||||
", c[0], c[1], c[2]));
|
||||
s.push_str(&format!("#define IGNEUM_BYTES_PER_HASH {}
|
||||
", p.bytes_per_hash()));
|
||||
s.push_str(&format!("#define IGNEUM_BYTES_PER_HASH {} // demand of the executed loads per nonce (loads x width); the memory moves sectors or lines
|
||||
", p.bytes_per_hash_executed()));
|
||||
s.push_str(&format!("#define IGNEUM_FOLD_ROT {FOLD_ROT}
|
||||
"));
|
||||
s.push_str(&format!("#define IGNEUM_FOLD_MUL {}
|
||||
|
|
@ -2016,7 +2016,7 @@ pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String {
|
|||
s.push_str("#define IGNEUM_LANES 32\n");
|
||||
s.push_str(&format!("#define IGNEUM_ITERATIONS {ITERATIONS}\n"));
|
||||
s.push_str(&format!("#define IGNEUM_INSTR_COUNT {INSTR_COUNT}\n"));
|
||||
s.push_str(&format!("#define IGNEUM_LOADS_PER_HASH {}\n", p.loads_per_hash()));
|
||||
s.push_str(&format!("#define IGNEUM_LOADS_PER_HASH {} // executed per nonce (twice the drawn program's under the 64-register window)\n", p.loads_per_hash_executed()));
|
||||
s.push_str(&format!("#define IGNEUM_WIDE_LOADS_PER_HASH {}\n", p.wide_loads_per_hash()));
|
||||
s.push_str(&format!("#define IGNEUM_OP_MIX {}\n", jstr(&p.op_mix())));
|
||||
s.push_str(&program_class_header_lines(p));
|
||||
|
|
@ -2258,7 +2258,7 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String {
|
|||
s.push_str(" \"registers\": 8,\n");
|
||||
s.push_str(&format!(" \"iterations\": {ITERATIONS},\n"));
|
||||
s.push_str(&format!(" \"instruction_count\": {INSTR_COUNT},\n"));
|
||||
s.push_str(&format!(" \"loads_per_hash\": {},\n", p.loads_per_hash()));
|
||||
s.push_str(&format!(" \"loads_per_hash\": {},\n", p.loads_per_hash_executed()));
|
||||
if p.program_class() != ProgramClass::V2 {
|
||||
s.push_str(&format!(" \"program_class\": {},\n", jstr(p.program_class().name())));
|
||||
if p.program_class() == ProgramClass::V4 {
|
||||
|
|
@ -2308,7 +2308,7 @@ pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String {
|
|||
s.push_str(&format!(" \"load_slots\": {},\n", p.class.load_slots));
|
||||
s.push_str(&format!(" \"load_mix_percent_4_16_64\": [{}, {}, {}],\n", p.class.mix[0], p.class.mix[1], p.class.mix[2]));
|
||||
s.push_str(&format!(" \"load_width_counts_4_16_64\": [{}, {}, {}],\n", c[0], c[1], c[2]));
|
||||
s.push_str(&format!(" \"bytes_per_hash\": {},\n", p.bytes_per_hash()));
|
||||
s.push_str(&format!(" \"bytes_per_hash\": {},\n", p.bytes_per_hash_executed()));
|
||||
if p.has_scratch() {
|
||||
s.push_str(&format!(" \"scratch_ops_per_hash\": {},\n", p.scratch_ops_per_hash()));
|
||||
s.push_str(&format!(" \"scratch_kib_per_warp\": {},\n", p.class.scratch_kb));
|
||||
|
|
|
|||
|
|
@ -1241,21 +1241,25 @@ impl Program {
|
|||
}
|
||||
out
|
||||
}
|
||||
/// Memory operations the hash EXECUTES per nonce (review B's V6-02, 8 October 2026): the scheduled statements
|
||||
/// (the drawn program's loads, twice under the 64-register window) times the eight iterations. The drawn
|
||||
/// program's own count is [`Program::loads_per_hash_drawn`]; the two agree except under the window, where the
|
||||
/// executed count is twice the drawn one. Every served loads figure names this schedule.
|
||||
/// The drawn program's memory operations per nonce (16 load slots times the eight iterations on every class):
|
||||
/// the count the acceptance rule and the draw's bookkeeping read (the acceptance judges the drawn program's
|
||||
/// sites). The hash EXECUTES [`Program::loads_per_hash_executed`] of them, twice this under the 64-register
|
||||
/// window; the served figures and the pack's headers name the executed count (review B's V6-02).
|
||||
pub fn loads_per_hash(&self) -> usize {
|
||||
let drawn = self.loads_per_hash_drawn();
|
||||
self.instrs.iter().filter(|i| i.op.is_load()).count() * ITERATIONS
|
||||
}
|
||||
|
||||
/// Memory operations the hash EXECUTES per nonce (review B's V6-02, 8 October 2026): the scheduled statements
|
||||
/// (the drawn program's loads, twice under the 64-register window) times the eight iterations; asserted to
|
||||
/// agree with the drawn count except by the window's factor of two. The emitters alone read it (program.h,
|
||||
/// program.json): the acceptance reads the drawn count, so no verdict moves with the served figure (the first
|
||||
/// form of this change, 0266c9ec0, routed the executed count into the acceptance and moved every reg64 verdict).
|
||||
pub fn loads_per_hash_executed(&self) -> usize {
|
||||
let drawn = self.loads_per_hash();
|
||||
let executed = self.scheduled().iter().filter(|i| i.op.is_load()).count() * ITERATIONS;
|
||||
assert!(executed == drawn || (self.class.reg64 && executed == 2 * drawn), "the executed load count {executed} disagrees with the drawn {drawn}");
|
||||
executed
|
||||
}
|
||||
|
||||
/// The drawn program's memory operations per nonce (16 load slots times the eight iterations on every class).
|
||||
pub fn loads_per_hash_drawn(&self) -> usize {
|
||||
self.instrs.iter().filter(|i| i.op.is_load()).count() * ITERATIONS
|
||||
}
|
||||
pub fn wide_loads_per_hash(&self) -> usize {
|
||||
self.instrs.iter().filter(|i| i.op == Op::WLoad).count() * ITERATIONS
|
||||
}
|
||||
|
|
@ -1263,11 +1267,17 @@ impl Program {
|
|||
self.instrs.iter().any(|i| i.op == Op::WLoad)
|
||||
}
|
||||
/// Dataset bytes read per hash: 4 per one-word load, 16 and 64 for the wider loads of the experiment.
|
||||
/// Dataset bytes the drawn program's loads demand per nonce (the acceptance's and the draw's figure).
|
||||
pub fn bytes_per_hash(&self) -> usize {
|
||||
self.instrs.iter().filter(|i| i.op == Op::Load).map(|i| i.width as usize * 4).sum::<usize>() * ITERATIONS
|
||||
}
|
||||
|
||||
/// Dataset bytes the hash's EXECUTED loads demand per nonce (review B's V6-02): the scheduled loads' widths in
|
||||
/// words times 4, times the eight iterations; a demand figure, not the memory's transactions (a 4-byte load is
|
||||
/// a 32-byte sector on NVIDIA and a 64-byte line on AMD; the served text names the model it quotes).
|
||||
pub fn bytes_per_hash(&self) -> usize {
|
||||
let drawn = self.instrs.iter().filter(|i| i.op == Op::Load).map(|i| i.width as usize * 4).sum::<usize>() * ITERATIONS;
|
||||
/// a 32-byte sector on NVIDIA and a 64-byte line on AMD; the served text names the model it quotes). The
|
||||
/// emitters alone read it.
|
||||
pub fn bytes_per_hash_executed(&self) -> usize {
|
||||
let drawn = self.bytes_per_hash();
|
||||
let executed = self.scheduled().iter().filter(|i| i.op == Op::Load).map(|i| i.width as usize * 4).sum::<usize>() * ITERATIONS;
|
||||
assert!(executed == drawn || (self.class.reg64 && executed == 2 * drawn), "the executed byte demand {executed} disagrees with the drawn {drawn}");
|
||||
executed
|
||||
|
|
|
|||
Loading…
Reference in a new issue