diff --git a/docs/analysis/cryptanalysis/report-acceptance-rule.md b/docs/analysis/cryptanalysis/report-acceptance-rule.md
index 9b51b13ee..ca83cdc54 100644
--- a/docs/analysis/cryptanalysis/report-acceptance-rule.md
+++ b/docs/analysis/cryptanalysis/report-acceptance-rule.md
@@ -255,6 +255,13 @@ it folds in act as fresh randomness on both sides: the closed form and the live
the 50-program widening is running (gap-50.log). Also: the cheap 256-unit proxy did rank real tail programs,
and 100064 sits 0.0032 above the 0.98 floor at 2^20, so the floor is live in the population tail, not idle.
+### Rule change 19:55 BST (box-hours honesty)
+
+The bounded class became 88 cores per box in total across all lanes: no new sweep starts except under the
+per-box lock `flock /srv/builds/_adv/locks/sweep.lock`, one sweep per box at a time; the running shards 00
+and 01 finish as they are; my adv-live confirmations and the row-90 census count against the ceiling, so
+nothing new starts until they end. Shards 02 to 09, the class-v5 check and any re-run go through the lock.
+
### Live hot-set census (RUNNING)
- box 2, 19:5x BST: `adv-live census --programs 2..201 --nonces 1000000 --threads 32` (200 consecutive accepted
diff --git a/tools/attack/adv-accept-v5/Cargo.lock b/tools/attack/adv-accept-v5/Cargo.lock
new file mode 100644
index 000000000..dbcec7322
--- /dev/null
+++ b/tools/attack/adv-accept-v5/Cargo.lock
@@ -0,0 +1,112 @@
+# This file is automatically @generated by Cargo.
+# It is not intended for manual editing.
+version = 4
+
+[[package]]
+name = "adv-accept-v5"
+version = "0.1.0"
+dependencies = [
+ "igneum-pow",
+]
+
+[[package]]
+name = "igneum-pow"
+version = "0.2.0"
+dependencies = [
+ "serde_json",
+]
+
+[[package]]
+name = "itoa"
+version = "1.0.18"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
+
+[[package]]
+name = "memchr"
+version = "2.8.3"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
+
+[[package]]
+name = "proc-macro2"
+version = "1.0.107"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
+dependencies = [
+ "unicode-ident",
+]
+
+[[package]]
+name = "quote"
+version = "1.0.47"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
+dependencies = [
+ "proc-macro2",
+]
+
+[[package]]
+name = "serde"
+version = "1.0.229"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
+dependencies = [
+ "serde_core",
+]
+
+[[package]]
+name = "serde_core"
+version = "1.0.229"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
+dependencies = [
+ "serde_derive",
+]
+
+[[package]]
+name = "serde_derive"
+version = "1.0.229"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
+dependencies = [
+ "proc-macro2",
+ "quote",
+ "syn",
+]
+
+[[package]]
+name = "serde_json"
+version = "1.0.151"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
+dependencies = [
+ "itoa",
+ "memchr",
+ "serde",
+ "serde_core",
+ "zmij",
+]
+
+[[package]]
+name = "syn"
+version = "3.0.6"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
+dependencies = [
+ "proc-macro2",
+ "quote",
+ "unicode-ident",
+]
+
+[[package]]
+name = "unicode-ident"
+version = "1.0.26"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
+
+[[package]]
+name = "zmij"
+version = "1.0.23"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
diff --git a/tools/attack/adv-accept-v5/Cargo.toml b/tools/attack/adv-accept-v5/Cargo.toml
new file mode 100644
index 000000000..49f9e9561
--- /dev/null
+++ b/tools/attack/adv-accept-v5/Cargo.toml
@@ -0,0 +1,21 @@
+[package]
+name = "adv-accept-v5"
+version = "0.1.0"
+edition = "2021"
+description = "adv-accept lane: the f8 hot-set harness over class v5 (vendored igneum-pow of branch class-v5, the dataset keyed by the window's state leaves), to ask whether a class v4 finding still reads hot under class v5"
+license = "MIT"
+publish = false
+
+[[bin]]
+name = "adv-live-v5"
+path = "src/live.rs"
+
+[dependencies]
+igneum-pow = { path = "igneum-pow" }
+
+[workspace]
+
+[profile.release]
+opt-level = 3
+lto = true
+codegen-units = 1
diff --git a/tools/attack/adv-accept-v5/igneum-pow/.gitignore b/tools/attack/adv-accept-v5/igneum-pow/.gitignore
new file mode 100644
index 000000000..2f7896d1d
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/.gitignore
@@ -0,0 +1 @@
+target/
diff --git a/tools/attack/adv-accept-v5/igneum-pow/Cargo.lock b/tools/attack/adv-accept-v5/igneum-pow/Cargo.lock
new file mode 100644
index 000000000..82ec948df
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/Cargo.lock
@@ -0,0 +1,105 @@
+# This file is automatically @generated by Cargo.
+# It is not intended for manual editing.
+version = 4
+
+[[package]]
+name = "igneum-pow"
+version = "0.2.0"
+dependencies = [
+ "serde_json",
+]
+
+[[package]]
+name = "itoa"
+version = "1.0.18"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
+
+[[package]]
+name = "memchr"
+version = "2.8.3"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
+
+[[package]]
+name = "proc-macro2"
+version = "1.0.107"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
+dependencies = [
+ "unicode-ident",
+]
+
+[[package]]
+name = "quote"
+version = "1.0.47"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
+dependencies = [
+ "proc-macro2",
+]
+
+[[package]]
+name = "serde"
+version = "1.0.229"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
+dependencies = [
+ "serde_core",
+]
+
+[[package]]
+name = "serde_core"
+version = "1.0.229"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
+dependencies = [
+ "serde_derive",
+]
+
+[[package]]
+name = "serde_derive"
+version = "1.0.229"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
+dependencies = [
+ "proc-macro2",
+ "quote",
+ "syn",
+]
+
+[[package]]
+name = "serde_json"
+version = "1.0.151"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
+dependencies = [
+ "itoa",
+ "memchr",
+ "serde",
+ "serde_core",
+ "zmij",
+]
+
+[[package]]
+name = "syn"
+version = "3.0.6"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
+dependencies = [
+ "proc-macro2",
+ "quote",
+ "unicode-ident",
+]
+
+[[package]]
+name = "unicode-ident"
+version = "1.0.26"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
+
+[[package]]
+name = "zmij"
+version = "1.0.23"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
diff --git a/tools/attack/adv-accept-v5/igneum-pow/Cargo.toml b/tools/attack/adv-accept-v5/igneum-pow/Cargo.toml
new file mode 100644
index 000000000..a5f0f4b63
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/Cargo.toml
@@ -0,0 +1,30 @@
+[package]
+name = "igneum-pow"
+version = "0.2.0"
+edition = "2021"
+description = "Igneum random-program GPU proof-of-work: seed, program generator (version 2: fixed load count, fresh sources, acceptance rule), memory-hard dataset, CPU warp verifier and kernel emitters; the source of every program pack"
+license = "MIT"
+publish = false
+
+[lib]
+name = "igneum_pow"
+path = "src/lib.rs"
+
+[[bin]]
+name = "igneum-pow"
+path = "src/main.rs"
+
+[dependencies]
+
+[dev-dependencies]
+serde_json = "1"
+
+# The cache fill is 2^22 ChaCha12 blocks and the vector tests derive thousands of items.
+# Unoptimised builds would make `cargo test` take minutes, so the dev profile is optimised too.
+[profile.dev]
+opt-level = 3
+
+[profile.release]
+opt-level = 3
+lto = true
+codegen-units = 1
diff --git a/tools/attack/adv-accept-v5/igneum-pow/README.md b/tools/attack/adv-accept-v5/igneum-pow/README.md
new file mode 100644
index 000000000..06507cca4
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/README.md
@@ -0,0 +1,181 @@
+# igneum-pow
+
+The Igneum lottery hash in Rust: the generator, the acceptance rule, the memory-hard dataset, the CPU verifier and the
+kernel emitters. This is the crate the rusty-kaspa fork calls (`docs/fork-map.md`, rows a1 to a3) and, since
+4 October 2026, the source of every program pack in `proto-cuda/packs/`. No dependency outside the standard
+library; `serde_json` is a dev-dependency for reading the packs in the tests.
+
+Dates: 3 October 2026 (crate, bit-exact with the Swift prototype), 4 October 2026 (generator version 2 and the
+acceptance rule; every vector re-cut). Toolchain: rustc 1.99.0 via rustup (the Homebrew 1.69 on PATH is too old;
+use `~/.cargo/bin/cargo`). Crate version 0.2.0.
+
+## Modules
+
+| Module | What it is | Swift namesake |
+|---|---|---|
+| `seed` | 32-byte seed words from bytes (FNV-1a 64, four salts, finalised); `seed_words_from_bytes` is the boundary where the chain feeds the epoch seed; SplitMix64 | `seedWordsBytes`, `SplitMix64` |
+| `generator` | version 2: 16 load slots drawn first from instructions 1..63, fresh-source loads, the other 48 ops from the ten non-load weights; attempts `k = 0, 1, ...` of a seed; the program id. The retired version 1 generator stays as `generate_v1` for the census and the lever measurements | `generateProgramV2`, `candidateProgram`, `generateProgramV1` |
+| `accept` | the acceptance rule of spec 01 section 1.4.6: two static tests and the 64-unit dynamic test on the seed-keyed closed-form dataset | `acceptProgram` |
+| `memhard` | 256 MiB cache (2^16 chains of 64 ChaCha12 blocks), mixer parameters, 8-round item derivation with the 32 lanes interleaved, `MemhardCpu::fetch` | `cpuFillCache`, `MixParams`, `deriveItems`, `MemhardCPU` |
+| `verify` | the 32-lane warp interpreter, `DatasetMode::{ClosedForm, MemoryHard}`, `Epoch`, `hash_warp`, `verify_block` | `cpuWarp`, `DatasetSource` |
+| `emit` | Metal, CUDA and OpenCL source, program.h, memhard.h, vectors.h, program.json, vectors.json, the header-bound kernels, `export_pack` | `generateMSL`, `generateCUDA`, `generateOpenCL`, `exportPack` |
+| `bind` | header binding (spec 01 section 1.6): init words from `"igneum-block/" \|\| H \|\| nonce_hi_le32`, bound hash API on `Epoch`, the 256-bit pow mapping, the interim day seed bytes | `blockInitWords` |
+
+## Generator version 2 (4 October 2026)
+
+Adopted from `docs/analysis/weak-program-census-2026-10-03.md` (ledger M5 and M6). Three parts, all in this crate
+and mirrored in `proto-metal/main.swift` so the Metal worker derives the same program from the same seed:
+
+| Part | Rule | Where |
+|---|---|---|
+| G1, exact load count | 16 `load` instructions per program, a uniform 16-subset of slots 1..63 drawn first by partial Fisher-Yates over the program stream; the other 48 ops from `add 12, xor 10, mul 8, mad 8, shfl 8, rotl 7, sub 6, mulhi 6, rotr 6, or 4` (sum 75). 128 loads per hash, 4,096 items per unit | `generator::candidate_from_words` |
+| G2, fresh source | a load's source is drawn from the registers other than `dst` written by an earlier instruction and not read by a load since, so no load repeats an earlier load's address in the hash | same |
+| R, acceptance | (a) no load whose source is unwritten since the previous load from it, cyclically; (b) every register has an `add`, `sub`, `xor`, `mad`, `shfl` or `load` write; (c) 64 units at base nonces from `SplitMix64(FNV-1a-64("igneum-accept/" \|\| seed words LE))`, init words = seed words, closed-form dataset `dataset_elem(idx, S[0], S[1])` at 2^28 words: no constant register bit, no lane-constant load site in any unit, fewer than 164 saturated final values, every output bit within 136 of 1,024, distinct addresses above 245,760 over the 2,048 hashes | `accept::check` |
+| Attempts | a rejected candidate is replaced by `seed_words_from_bytes(seed \|\| k_le32)` for `k = 1, 2, ...`; 32 consecutive rejections are a consensus fault (probability below 2^-136 at the measured 5 percent rate) | `generator::generate_from_seed_bytes` |
+| Program id | `FNV-1a-64("igneum-program/" \|\| 2_le32 \|\| seed words LE \|\| attempt_le32)`, written into program.json and program.h with the generator version and the attempt, so a version 1 pack or another attempt can never pass for the current program | `Program::program_id` |
+
+Measured on this crate (`igneum-pow accept`): `igneum-genesis` attempt 0 accepted, program id `bcc1248b10cc90f2`,
+op mix `load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1`, 128.000 distinct addresses per
+hash; `igneum-hourly` attempt 0, id `a4c4d00961c855df`; seeds `igneum-census-2026-10-03/22`, `/37` and `/51` have
+attempt 0 rejected ((b) r7 without an injecting write; (c) 119.74 distinct addresses; (b) r4) and attempt 1
+accepted (ids `22ed0609d079f4cf`, `947705cc4eb1df0a`, `9869afcc028bf9f1`); those three are the conformance vectors
+for the attempt rule. The rule costs 1.3 to 3.4 ms per seed on one core. The 20,000-program census under this
+generator is in `docs/bench-log.md` (4 October 2026 entry).
+
+## The API the fork calls
+
+```rust
+use igneum_pow::{Epoch, DatasetMode};
+
+// Once per epoch and day: derives the accepted program and fills the 256 MiB cache (about 0.2 s on one core).
+let epoch = Epoch::memory_hard("igneum-genesis", "2026-10-03");
+
+let h: u64 = epoch.hash(nonce); // one nonce (computes its aligned 32-nonce warp)
+let w: [u64; 32] = epoch.hash_warp(base_nonce); // one warp
+let ok: bool = epoch.verify_block(nonce, target_u64);
+
+// Miner programs for the epoch: the pack every worker compiles.
+let pack = igneum_pow::emit::export_pack(&epoch, "2026-10-03", "igneum node");
+pack.write_to(std::path::Path::new("out"))?; // kernel.cu, kernel.cl, program.metal, memhard.h, ...
+```
+
+## The header-bound form (what the chain uses)
+
+The pack form above initialises the lane registers from the program's own seed words, so one nonce has one
+hash per epoch whatever block is mined. On the chain the init words commit to the block (spec 01 section 1.6,
+`src/bind.rs`):
+
+```
+H = header hash with the nonce field zeroed, every other field as mined
+ (rusty-kaspa hash_override_nonce_time(header, 0, header.timestamp))
+nonce = 64 bits; lane nonce n = low 32 bits; nonce_hi = high 32 bits
+I = seed_words_from_bytes("igneum-block/" || H || nonce_hi_le32) (49 bytes in)
+hash = interpret(program, I, n) (section 1.7 of the spec)
+pow256 = hash in the top 64 bits, low 192 bits zero (little-endian bytes 24..32)
+valid = pow256 <= target256, which is exactly hash <= target256 >> 192
+```
+
+`H` keeps the timestamp (Kaspa zeroes it in the pre-PoW hash and absorbs it in cSHAKE afterwards; the lane hash has
+no afterwards, so a nonce would otherwise be reusable across timestamps). The interim day seed is
+`"igneum-day/" || day_le64` with `day = timestamp_ms / 86,400,000`. The epoch seed bytes are the 32 bytes of the
+epoch block hash (devnet); the program is `generate_from_seed_bytes(epoch_seed)`, attempts included.
+
+```rust
+use igneum_pow::{bind, Epoch};
+
+let epoch = Epoch::from_seed_bytes(epoch_hash.as_bytes(), &bind::day_bytes(day), "label");
+let lane: u64 = epoch.hash_bound(&prehash, nonce); // one 64-bit nonce
+let warp: [u64; 32] = epoch.hash_warp_bound(&prehash, nonce); // its aligned 32-nonce group
+let init = bind::block_init_words(&prehash, nonce); // what a GPU kernel takes as its argument
+let same = epoch.hash_warp_init(&init, nonce as u32 & !31); // == warp
+let pow: [u8; 32] = epoch.pow_bound(&prehash, nonce);
+let ok = epoch.verify_block_bound(&prehash, nonce, bind::target64_from_le256(&target_le));
+```
+
+On the GPU the init words are a kernel argument: `igneum_hash_bound` in `program_bound.metal` takes
+`constant uint* initw [[buffer(3)]]`, `kernel_bound.cu` takes `IgneumInitWords iw` by value, `kernel_bound.cl` an
+`initw` buffer. All three differ from `igneum_hash` only in the kernel name, the argument and the eight init lines.
+
+Bound vectors (seed `igneum-genesis`, day `2026-10-03`, memory-hard, 2^28 words, generator v2; `igneum-pow hash-bound`):
+
+| H | nonce | hash_bound |
+|---|---|---|
+| 32 zero bytes | 0 | `746c567b090acf6a` |
+| 32 zero bytes | 1 | `45a619f860880c73` |
+| 32 zero bytes | 31 | `9aa495e43dedbfe6` |
+| 32 zero bytes | 4096 | `2e6ffd7624d3cba2` |
+| 32 zero bytes | 4294967296 (1 << 32) | `38a5cea1fb01431a` |
+| bytes 00 01 02 .. 1f | 0 | `2a79c5e4797bf6aa` |
+| bytes 00 01 02 .. 1f | 4294967301 ((1 << 32) + 5) | `a243e0c61aa1b82e` |
+| bytes 00 01 02 .. 1f | 18446744073709551615 (u64::MAX) | `9c7bbfbd064fe1a4` |
+
+Init words for H = 32 zero bytes, nonce 0: `595a8f8a 37647e95 faadade1 cbbcf2a4 54f7cc13 f6851b5e 8c68ca04 7991ea9c`
+(unchanged by v2: the binding does not depend on the program). The eight are pinned in `bind::tests::bound_vectors`.
+The version 1 bound vectors of 3 October 2026 are retired.
+
+`Epoch` is `Send + Sync`; build one and share it. `DatasetMode::ClosedForm` is the interpreter regression dataset
+of the packs `igneum-genesis` and `igneum-hourly` and is not memory-hard.
+
+## CLI
+
+```
+cargo build --release
+./target/release/igneum-pow bench --seed igneum-genesis [--warps 20] [--closed-form] [--day 2026-10-03]
+./target/release/igneum-pow export --seed igneum-genesis --out
[--closed-form]
+./target/release/igneum-pow export --epoch-hex <64 hex> --day-hex --out the chain's byte seeds
+./target/release/igneum-pow hash --seed igneum-genesis --nonce 4103
+./target/release/igneum-pow hash-bound --seed igneum-genesis --prehash <64 hex> --nonce
+./target/release/igneum-pow accept --seed igneum-census-2026-10-03/22 every candidate with its verdict
+./target/release/igneum-pow show --seed igneum-genesis the accepted program, one line per instruction
+```
+
+The packs were regenerated on 4 October 2026 with exactly these commands:
+
+```
+igneum-pow export --closed-form --seed igneum-genesis --out ../proto-cuda/packs/igneum-genesis
+igneum-pow export --closed-form --seed igneum-hourly --out ../proto-cuda/packs/igneum-hourly
+igneum-pow export --seed igneum-genesis --out ../proto-cuda/packs/igneum-genesis-mh
+igneum-pow export --epoch-hex edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07 \
+ --day-hex 69676e65756d2d6461792ffa50000000000000 --out ../proto-cuda/packs/igneum-devnet-v4-epoch0
+```
+
+The last is the chain's own derivation for devnet v4 epoch 0: the devnet genesis hash as the epoch seed and
+`bind::day_bytes(20730)` (2026-10-04) as the day bytes.
+
+## Tests
+
+`cargo test --release` (39 tests, about 2 s after compile; the dev profile is optimised so the cache fill is quick):
+
+| Check | Pack | Result |
+|---|---|---|
+| program.json instruction by instruction from `seed_bytes`, generator 2, attempt, program id, op mix, 128 loads, acceptance | all four | match |
+| Mixer parameters (key, rot, mul, rc) | igneum-genesis-mh, igneum-devnet-v4-epoch0 | match |
+| Cache head, last line, FNV-1a 64 (`48c4f5bf24166b2e` for day 2026-10-03, `448274a57f508cbc` for day bytes 20730) | the two memory-hard packs | match |
+| Dataset head (16), `[MASK]`, 64 sampled words | all four | match |
+| 96 hash vectors (3 warps x 32 lanes) | all four | 96/96 each |
+| kernel.cu, kernel_bound.cu, program.metal, program_bound.metal, kernel.cl, kernel_bound.cl, program.h, program.json byte-identical; 16 masked loads per kernel | all four | identical |
+| memhard.h, memhard.metal byte-identical | the two memory-hard packs | identical |
+| vectors.json, vectors.h byte-identical; no stale file in any pack directory | all four | identical |
+| bound vectors (8), bound warp == bound single, H and nonce_hi enter the hash | igneum-genesis-mh | pass |
+| the devnet pack equals `Epoch::from_seed_bytes(genesis, day_bytes(20730))` | igneum-devnet-v4-epoch0 | pass |
+| acceptance: instrumented interpreter == `hash_warp`, cyclic stale-load detection, injecting-write detection, rejection under 12.5 percent and distinct loads above 127 on 400 census seeds, version 1 programs mostly rejected | unit tests | pass |
+| generator: 16 loads, none at instruction 0, contract on 200 candidates; fresh sources; attempt words; program ids separate versions and attempts | unit tests | pass |
+
+## Measured, Apple M5 Max, one core, release build
+
+| Step | 3 October 2026 (v1, 104 loads) | 4 October 2026 (v2, 128 loads, 4,096 items) |
+|---|---|---|
+| Cache fill, 256 MiB | 175 to 181 ms | 179 ms |
+| CPU verify per warp, igneum-genesis, avg of 20 | 0.441 ms (3,328 items) | 0.631 ms |
+| Cold single warps (bases 0, 4096, 1000000) | 0.41 to 0.87 ms | 0.67 to 0.81 ms |
+| Acceptance rule per candidate | | 1.3 to 3.4 ms |
+| Closed form, igneum-genesis | 0.002 ms | 0.002 ms |
+
+The 10 ms gate holds with a margin of about 16x steady on this core. Every verified unit now derives exactly 4,096
+items, the design bound of spec section 1.11.
+
+## Not done here
+
+- No GPU. The vectors tie this crate to the Metal, CUDA and OpenCL results through the packs; nothing here runs a kernel.
+- The epoch seed enters at `Epoch::from_seed_bytes` (the epoch block hash on devnet); the VDF output replaces it there.
+- The dynamic acceptance test is specified on the closed-form dataset at 2^28 words; if the prototype mask ever changes, the rule's constant stays at 2^28.
diff --git a/tools/attack/adv-accept-v5/igneum-pow/VENDORED-FROM.txt b/tools/attack/adv-accept-v5/igneum-pow/VENDORED-FROM.txt
new file mode 100644
index 000000000..76140bcb6
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/VENDORED-FROM.txt
@@ -0,0 +1 @@
+class-v5 igneum-pow at commit 25c8063ff38e248b7a92d91575e975a81d953894
diff --git a/tools/attack/adv-accept-v5/igneum-pow/rustfmt.toml b/tools/attack/adv-accept-v5/igneum-pow/rustfmt.toml
new file mode 100644
index 000000000..c775577ee
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/rustfmt.toml
@@ -0,0 +1,2 @@
+max_width = 120
+use_small_heuristics = "Max"
diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/accept.rs b/tools/attack/adv-accept-v5/igneum-pow/src/accept.rs
new file mode 100644
index 000000000..709d6f432
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/src/accept.rs
@@ -0,0 +1,869 @@
+//! Program acceptance (spec 01 section 1.4.6, adopted 4 October 2026 from the weak-program census of
+//! `docs/analysis/weak-program-census-2026-10-03.md`, section 6).
+//!
+//! A candidate program is accepted only if every test below holds. Every conforming implementation evaluates
+//! exactly these tests on exactly these inputs, so every node skips the same seeds.
+//!
+//! | Part | Test |
+//! |---|---|
+//! | (a) static | for every `load`, some instruction between the previous `load` from the same source register and this one, in cyclic order over the 64 instructions, writes that register |
+//! | (b) static | every register `r0..r7` is the destination of at least one `add`, `sub`, `xor`, `mad`, `shfl` or `load` |
+//! | (c) dynamic | the program is interpreted for [`ACCEPT_UNITS`] (64) units of 32 lanes at base nonces drawn from SplitMix64 seeded with `FNV-1a-64("igneum-accept/" \|\| seed words as little-endian bytes)`, each `low32(next()) AND NOT 31`, with init words equal to the seed words and the closed-form dataset `dataset_elem(idx, S[0], S[1])` at [`ACCEPT_DATASET_LOG2`] (2^28 words) in place of the memory-hard dataset. Over the 2,048 evaluations: no register has a bit equal in every final value; no load site (iteration, instruction) reads one address in all 32 lanes of any unit; fewer than [`MAX_SATURATED`] (164, 1 percent of 16,384) final register values are 0 or 2^32 - 1; every output bit's ones count is within [`BIAS_TOLERANCE`] (136, 6 sigma) of 1,024; the distinct masked addresses read by one lane in one evaluation, summed over the 2,048 evaluations, exceed [`MIN_DISTINCT_SUM`] (245,760, a mean above 120 of the 128 loads) |
+//!
+//! The dynamic test uses the closed form so that it is a pure function of the program (no cache, no day) and
+//! costs about a millisecond on one core. A hot-table load (`docs/plans/hot-table.md`) reads the closed form keyed by
+//! seed words 2 and 3 at its multiply-shift index, a second pure table beside the dataset stand-in (words 0 and 1). The census (section 7.3) checked on 100,000 programs that the
+//! closed-form verdict agrees with the memory-hard one on all but 39 threshold-edge cases.
+
+use crate::generator::{Instr, LoadClass, Op, Program, ShadowClass, INSTR_COUNT, ITERATIONS, LANES, V4_CLASS, V4_SHADOW_INSTRS};
+use crate::seed::{fnv1a64, SplitMix64};
+use crate::memhard::hot_index;
+use crate::verify::{dataset_elem, fold_words, load_index, splitmix32, ScratchModel};
+
+/// Units (32-lane warps) the dynamic test interprets.
+pub const ACCEPT_UNITS: usize = 64;
+
+/// Class v4 sub-version 3, rule (c''): the per-site distinct-index RATIO (AP-F8-1's low-entropy-band class, 7 October
+/// 2026, the attack-pass gate's numbers through main), keyed on the class v4 shape. Over [`ACCEPT_UNITS_DISTINCT_V4`]
+/// units (2^20 evaluations per site) the count of distinct dataset word indices a load site reads, against the
+/// expectation of a uniform source on the site's window (N - N^2 / 2W), must reach [`MIN_DISTINCT_RATIO_V4`]. The
+/// floor sits between the 55 clean F8 seeds' minimum over their site rows (0.9960; the clean p1 0.9990, the median
+/// 1.0000) and the strong failing seeds' maximum (p56 0.9654; p23 0.8361, p18 0.9274, p19 0.9335, p15 0.9432),
+/// 0.015 from each. The open tail: F8's p4, p8, p10 and p34 (1.22x to 1.50x on the gate) read 0.9927 to 0.9963 at
+/// 2^20, inside the clean spread, and a 2^24 pass does not separate them either (p34 0.9181, p4 0.9614, p8 0.9630,
+/// p10 0.9612 against the clean p44 0.9612, p52 0.9613, p3 0.9971, p2 and p5 1.0004); they stay unattributed and
+/// chased in `docs/fud-ledger.md` AP-F8-1. A candidate under the floor is rejected and the next attempt drawn under
+/// the 256 cap and the last resort. Measured on one box-2 core with the shadow executed: 2.8 s per chosen candidate.
+pub const ACCEPT_UNITS_DISTINCT_V4: usize = 4096;
+/// The ratio floor at 2^20.
+pub const MIN_DISTINCT_RATIO_V4: f64 = 0.98;
+/// Kept for the record and the driver, not wired: the most repeated source value per site over the (c) units'
+/// 16,384 evaluations (a uniform site repeats a value 2 or 3 times; the finding's bands sit under the ratio instead).
+pub const MAX_SOURCE_REPEAT_V4: u32 = 8;
+/// Hashes the dynamic test evaluates: 2,048.
+pub const ACCEPT_HASHES: usize = ACCEPT_UNITS * LANES;
+/// Domain tag of the base-nonce stream.
+pub const ACCEPT_TAG: &[u8] = b"igneum-accept/";
+/// log2 of the closed-form dataset the test addresses: the prototype's 2^28 words, MASK 0x0fffffff.
+pub const ACCEPT_DATASET_LOG2: u32 = 28;
+/// Final register values equal to 0 or 2^32 - 1 must number fewer than this (1 percent of 8 x 2,048).
+pub const MAX_SATURATED: u32 = 164;
+/// Every output bit's ones count must be within this of 1,024 (6 x sqrt(2048) / 2, rounded).
+pub const BIAS_TOLERANCE: u32 = 136;
+/// Distinct addresses per lane per evaluation, summed over 2,048 evaluations, must exceed this (mean above 120).
+pub const MIN_DISTINCT_SUM: u64 = 245_760;
+
+/// The distinct-address bound for a program with `loads` dataset loads per hash: the same 120 of 128 ratio, so
+/// [`MIN_DISTINCT_SUM`] for the lottery hash and `loads x 1,920` for the read-width classes with other counts.
+/// Variant 5's scratch read-modify-writes are not dataset loads: their slots repeat by design (a later
+/// read-modify-write sees an earlier write), so they are neither counted nor bounded here.
+pub fn min_distinct_sum(loads: usize) -> u64 {
+ loads as u64 * ACCEPT_HASHES as u64 * 120 / 128
+}
+
+/// Why a candidate was rejected. The verdict (accept or reject) is what consensus depends on; the reason is the
+/// first failing test in the order of the module table.
+#[derive(Clone, Copy, Debug, PartialEq, Eq)]
+pub enum Reject {
+ /// (a): instruction `instr` loads from `reg`, which no instruction wrote since the previous load from it.
+ StaleLoadSource { instr: u8, reg: u8 },
+ /// (b): no injecting op writes `reg`.
+ NoInjectingWrite { reg: u8 },
+ /// (c): `reg` has `bits` bits equal in all 2,048 final values.
+ ConstantBit { reg: u8, bits: u8 },
+ /// (c): the load at `instr` in `iteration` read one address in all 32 lanes of `unit`.
+ LaneConstantSite { iteration: u8, instr: u8, unit: u8 },
+ /// (c): `count` final register values were 0 or all ones.
+ Saturated { count: u32 },
+ /// (a'), class v4 sub-version 2 (AP-F8-1): the load at `instr` reads `reg`, which is not fresh by dataflow in the
+ /// steady state of the loop (the freshness fixpoint over the base program and the shadow block).
+ UnfreshLoadSource { instr: u8, reg: u8 },
+ /// (c'), class v4 sub-version 2 (AP-F8-1): the load at `site` read a source value of 0 or all ones in `count` of
+ /// its 16,384 evaluations (64 units x 32 lanes x 8 iterations); limit [`MAX_SATURATED`] - 1, the same 1 percent as (c).
+ SaturatedSource { site: u8, count: u32 },
+ /// (c'') (B), class v4 sub-version 3: the load at `site` read the value `value` in `count` of its 16,384 (c)
+ /// evaluations (limit [`MAX_SOURCE_REPEAT_V4`] - 1): one constant upstream that the lineage rule cannot see.
+ RepeatedSource { site: u8, value: u32, count: u32 },
+ /// (c''), class v4 sub-version 3: the load at `site` read `distinct` distinct dataset word indices over
+ /// `evaluations`, `ratio_milli` / 1000 of a uniform source on its window, under the floor: a low-entropy index band
+ /// (F8's p23, p18, p19, p15, p56).
+ LowEntropySite { site: u8, distinct: u32, evaluations: u32, ratio_milli: u32 },
+ /// (c): output bit `bit` was set in `ones` of 2,048 hashes.
+ OutputBias { bit: u8, ones: u32 },
+ /// (c): the distinct-address sum was `sum`.
+ DistinctAddresses { sum: u64 },
+}
+
+impl std::fmt::Display for Reject {
+ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+ match self {
+ Reject::StaleLoadSource { instr, reg } => {
+ write!(f, "(a) load at instruction {instr} reads r{reg}, unwritten since the previous load from it")
+ }
+ Reject::NoInjectingWrite { reg } => write!(f, "(b) r{reg} has no add, sub, xor, mad, shfl or load write"),
+ Reject::ConstantBit { reg, bits } => write!(f, "(c) r{reg} has {bits} nonce-independent bits"),
+ Reject::LaneConstantSite { iteration, instr, unit } => {
+ write!(f, "(c) load at iteration {iteration} instruction {instr} reads one address in all lanes of unit {unit}")
+ }
+ Reject::Saturated { count } => write!(f, "(c) {count} of 16384 final register values saturated (limit 163)"),
+ Reject::UnfreshLoadSource { instr, reg } => write!(f, "(a') load at {instr} reads r{reg}, not fresh by dataflow in the loop's steady state (class v4 sub-version 2)"),
+ Reject::RepeatedSource { site, value, count } => write!(f, "(c'') load site {site} read the value {value:#010x} in {count} of 16384 evaluations (limit {})", MAX_SOURCE_REPEAT_V4 - 1),
+ Reject::LowEntropySite { site, distinct, evaluations, ratio_milli } => write!(f, "(c'') load site {site} read {distinct} distinct word indices over {evaluations} evaluations, {}.{:03} of a uniform source on its window (floor {MIN_DISTINCT_RATIO_V4} at 2^20)", ratio_milli / 1000, ratio_milli % 1000),
+ Reject::SaturatedSource { site, count } => write!(f, "(c') load site {site} read a saturated source value in {count} of 16384 evaluations (limit 163)"),
+ Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"),
+ Reject::DistinctAddresses { sum } => {
+ write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0)
+ }
+ }
+ }
+}
+
+/// What the dynamic test measured on an accepted program.
+#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
+pub struct AcceptReport {
+ /// Distinct masked addresses per lane per evaluation, summed over the 2,048 evaluations.
+ pub distinct_sum: u64,
+ /// Final register values equal to 0 or all ones.
+ pub saturated: u32,
+ /// The largest `|ones - 1024|` over the 64 output bits.
+ pub bias_max: u32,
+}
+
+impl AcceptReport {
+ /// Mean distinct addresses per hash (128 at most).
+ pub fn distinct_mean(&self) -> f64 {
+ self.distinct_sum as f64 / ACCEPT_HASHES as f64
+ }
+}
+
+/// Part (a): no load whose source is unwritten since the previous load from it, cyclically.
+fn check_stale_loads(instrs: &[Instr]) -> Result<(), Reject> {
+ // `pending[r]`: a load has read r and nothing has written r since. Two passes over the list so the second
+ // pass sees the state carried over the iteration boundary.
+ let mut pending = [false; 8];
+ for _pass in 0..2 {
+ for (k, ins) in instrs.iter().enumerate() {
+ if ins.op.is_load() && pending[ins.src as usize] {
+ return Err(Reject::StaleLoadSource { instr: k as u8, reg: ins.src });
+ }
+ pending[ins.dst as usize] = false;
+ if ins.op.is_load() {
+ pending[ins.src as usize] = true;
+ }
+ }
+ }
+ Ok(())
+}
+
+/// Part (b): every register has an injecting write.
+fn check_injecting_writes(instrs: &[Instr]) -> Result<(), Reject> {
+ let mut injected = [false; 8];
+ for ins in instrs {
+ if ins.op.injects() {
+ injected[ins.dst as usize] = true;
+ }
+ }
+ for (reg, ok) in injected.iter().enumerate() {
+ if !ok {
+ return Err(Reject::NoInjectingWrite { reg: reg as u8 });
+ }
+ }
+ Ok(())
+}
+
+/// The most repeated value of `values` (sorted in place) and that value: (B), kept for the driver, not wired.
+#[allow(dead_code)]
+fn most_repeated(values: &mut [u32]) -> (u32, u32) {
+ values.sort_unstable();
+ let (mut best, mut best_v, mut run) = (0u32, 0u32, 0u32);
+ for i in 0..values.len() {
+ run = if i > 0 && values[i] == values[i - 1] { run + 1 } else { 1 };
+ if run > best {
+ best = run;
+ best_v = values[i];
+ }
+ }
+ (best, best_v)
+}
+
+/// (c''), class v4 sub-version 3: the 2^20 ratio pass on the chosen candidate (the constants above).
+pub fn check_distinct_indices_v4(p: &Program) -> Result<(), Reject> {
+ distinct_ratio_pass(p, ACCEPT_UNITS_DISTINCT_V4, MIN_DISTINCT_RATIO_V4).map(|_| ())
+}
+
+/// One ratio pass over `units`: every load site's distinct word indices against the uniform expectation on its
+/// window (`N - N^2 / 2W`, the window `2^28 >> min(win, 2)` words of the closed-form dataset), `Err` at the first
+/// site under `floor`, else the minimum ratio and its site.
+pub fn distinct_ratio_pass(p: &Program, units: usize, floor: f64) -> Result<(f64, usize), Reject> {
+ let n = (units * LANES * ITERATIONS) as f64;
+ let d = distinct_indices_v4(p, units)?;
+ let mut min = (f64::MAX, 0usize);
+ let mut site = 0usize;
+ for i in &p.instrs {
+ if !i.op.is_load() {
+ continue;
+ }
+ let wsize = ((1u64 << ACCEPT_DATASET_LOG2) >> (i.win as u64).min(2)) as f64;
+ let ratio = d[site] as f64 / (n - n * n / (2.0 * wsize));
+ if ratio < floor {
+ return Err(Reject::LowEntropySite { site: site as u8, distinct: d[site], evaluations: n as u32, ratio_milli: (ratio * 1000.0) as u32 });
+ }
+ if ratio < min.0 {
+ min = (ratio, site);
+ }
+ site += 1;
+ }
+ Ok(min)
+}
+
+/// The distinct dataset word indices every load site reads over `units` units of the seed's acceptance stream on
+/// the closed-form words (the sample caps the count near `units x 32 x 8`, so a site's index entropy is read only
+/// below about log2 of that).
+pub fn distinct_indices_v4(p: &Program, units: usize) -> Result, Reject> {
+ let loads = p.loads_per_hash();
+ let sites = loads / ITERATIONS;
+ let mut acc = Acc {
+ sources: None,
+ indices: Some(vec![Vec::with_capacity(units * LANES * ITERATIONS); sites]),
+ sat_source: vec![0; sites],
+ and_acc: [u32::MAX; 8],
+ or_acc: [0; 8],
+ saturated: 0,
+ bit_ones: [0; 64],
+ distinct_sum: 0,
+ };
+ let mut lane_addrs = vec![0u32; LANES * loads];
+ for (unit, &base) in accept_base_nonces_n(&p.seed, units).iter().enumerate() {
+ run_unit(p, unit, base, &mut acc, &mut lane_addrs)?;
+ }
+ let mut out = Vec::with_capacity(sites);
+ for ix in acc.indices.take().unwrap().iter_mut() {
+ ix.sort_unstable();
+ ix.dedup();
+ out.push(ix.len() as u32);
+ }
+ Ok(out)
+}
+
+/// Whether `class` is the class v4 shape (the 256-instruction shadow block over the class v3 base, the pass count and
+/// the era set aside): the shape the sub-version 2 rules (a') and (c') apply to, on every draw path.
+pub fn is_class_v4_shape(class: &LoadClass) -> bool {
+ matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. }))
+ // class v5 (docs/design/class-v5-stored-state.md) is judged under the same rules: its state flag is set aside
+ && LoadClass { era: None, shadow: None, state: false, ..*class } == LoadClass { shadow: None, ..V4_CLASS }
+}
+
+/// One pass of the dataflow freshness over the base program then the shadow block (the order of one iteration),
+/// from `fresh`; `check` reports the first load that reads a register that is not fresh. The rule (AP-F8-1,
+/// `docs/analysis/ca3-v4-uniform.md`): a load leaves its destination fresh only if its source was (a saturated
+/// source reads one fixed word); add, sub, xor, mad and shfl if either operand was; rotl and rotr if the operand
+/// was (a rotate maps all-ones and zero to themselves); or, mul and mulhi never.
+fn freshness_pass(p: &Program, fresh: &mut [bool; 8], pair_op: &mut [Option<(Op, usize)>; 8], check: bool) -> Result<(), Reject> {
+ for (k, i) in p.instrs.iter().chain(p.shadow.iter()).enumerate() {
+ let (d, a) = (i.dst as usize, i.src as usize);
+ if check && i.op.is_load() && !fresh[a] {
+ return Err(Reject::UnfreshLoadSource { instr: k as u8, reg: i.src });
+ }
+ // the shared-operand idiom (sub-version 3): or-then-xor or or-then-sub on one operand is `d & ~s`, xor-then-or
+ // is `d | s`: lossy, though the second op would inject on its own (F8's p23: `or r6 |= r4; xor r6 ^= r4`)
+ let masked = matches!((pair_op[d], i.op), (Some((Op::Or, s)), Op::Xor) | (Some((Op::Or, s)), Op::Sub) | (Some((Op::Xor, s)), Op::Or) if s == a);
+ fresh[d] = !masked
+ && match i.op {
+ Op::Load | Op::WLoad | Op::Scratch | Op::Hot => fresh[a],
+ Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl => fresh[d] || fresh[a],
+ Op::Rotl | Op::Rotr => fresh[d],
+ Op::Or | Op::Mul | Op::MulHi => false,
+ };
+ pair_op[d] = if matches!(i.op, Op::Or | Op::Xor) && !masked { Some((i.op, a)) } else { None };
+ for r in 0..8 {
+ if r != d {
+ if let Some((_, s)) = pair_op[r] {
+ if s == d {
+ pair_op[r] = None;
+ }
+ }
+ }
+ }
+ }
+ Ok(())
+}
+
+/// Part (a'), class v4 sub-version 2: every load's source is fresh by dataflow in the loop's steady state. The draw
+/// of `candidate_from_words_class` keeps in-pass sources fresh; this closes the iteration boundary (a source last
+/// written late in the previous iteration or in the shadow block, which the draw's no-eligible fallback can pick:
+/// F8's p11, an `or` at 63 feeding a load at 1). The state starts all fresh (the init words are a per-lane hash of
+/// the nonce) and is run to its fixpoint (it only ever falls, so at most 8 passes change it), then one checking pass.
+pub fn check_fresh_sources_v4(p: &Program) -> Result<(), Reject> {
+ if !is_class_v4_shape(&p.class) {
+ return Ok(());
+ }
+ let mut fresh = [true; 8];
+ let mut pair_op: [Option<(Op, usize)>; 8] = [None; 8];
+ for _ in 0..9 {
+ let before = (fresh, pair_op);
+ freshness_pass(p, &mut fresh, &mut pair_op, false)?;
+ if (fresh, pair_op) == before {
+ break;
+ }
+ }
+ freshness_pass(p, &mut fresh, &mut pair_op, true)
+}
+
+/// Parts (a), (b) and, for class v4 sub-version 2, (a').
+pub fn check_static(p: &Program) -> Result<(), Reject> {
+ if p.instrs.len() != INSTR_COUNT {
+ panic!("acceptance needs a {INSTR_COUNT}-instruction program");
+ }
+ check_stale_loads(&p.instrs)?;
+ check_injecting_writes(&p.instrs)?;
+ check_fresh_sources_v4(p)
+}
+
+/// The 64 base nonces of the dynamic test for seed words `seed`.
+pub fn accept_base_nonces(seed: &[u32; 8]) -> [u32; ACCEPT_UNITS] {
+ let v = accept_base_nonces_n(seed, ACCEPT_UNITS);
+ let mut out = [0u32; ACCEPT_UNITS];
+ out.copy_from_slice(&v);
+ out
+}
+
+/// The first `n` base nonces of the seed's acceptance stream (the (c) units are the first [`ACCEPT_UNITS`]).
+pub fn accept_base_nonces_n(seed: &[u32; 8], n: usize) -> Vec {
+ let mut b = Vec::with_capacity(ACCEPT_TAG.len() + 32);
+ b.extend_from_slice(ACCEPT_TAG);
+ for w in seed {
+ b.extend_from_slice(&w.to_le_bytes());
+ }
+ let mut rng = SplitMix64::new(fnv1a64(&b));
+ (0..n).map(|_| (rng.next() as u32) & !31).collect()
+}
+
+#[inline(always)]
+fn mulhi32(a: u32, b: u32) -> u32 {
+ ((a as u64 * b as u64) >> 32) as u32
+}
+
+/// Accumulators of the dynamic test over the 64 units.
+struct Acc {
+ /// (c'') (B): every load's source value per site, recorded when present.
+ sources: Option>>,
+ /// (c'') (A): every load's dataset index per site, recorded when present (the distinct-index pass only).
+ indices: Option>>,
+ /// (c'): per load site (the load's index within the iteration), how many of its evaluations read a source value
+ /// of 0 or all ones (class v4 sub-version 2; counted for every class, judged for class v4 only).
+ sat_source: Vec,
+ and_acc: [u32; 8],
+ or_acc: [u32; 8],
+ saturated: u32,
+ bit_ones: [u32; 64],
+ distinct_sum: u64,
+}
+
+/// One unit of the dynamic test: the interpreter of `verify.rs` with the closed-form dataset, instrumented.
+/// Returns the first lane-constant load site, if any.
+fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut [u32]) -> Result<(), Reject> {
+ let seed = &p.seed;
+ let mask: u32 = (1u32 << ACCEPT_DATASET_LOG2) - 1;
+ let (d0, d1) = (seed[0], seed[1]);
+ let (h0, h1) = (seed[2], seed[3]);
+ let hot_words = p.hot_words();
+ let loads = p.loads_per_hash();
+ let mut r = [[0u32; LANES]; 8];
+ for lane in 0..LANES {
+ let nonce = base.wrapping_add(lane as u32);
+ for i in 0..8 {
+ let mut x = nonce ^ seed[i];
+ x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
+ x = splitmix32(x);
+ r[i][lane] = x ^ seed[(i + 1) & 7];
+ }
+ }
+ let mut idx = [0u32; LANES];
+ let mut nload = 0usize;
+ let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None };
+ let slot_mask = p.class.scratch_slot_mask();
+ let era = p.class.era;
+ // Class v4 sub-version 3 (AP-F8-3, 7 October 2026): the acceptance interpreter runs the latency-shadow block
+ // after instruction 63 of every iteration, `reps` times with the iteration's `sel`, exactly as the hash does
+ // (verify.rs). Until this commit it ran the 64 base instructions only, so every dynamic test (c) judged a class v4
+ // program the chain never hashes. The shadow block holds no load, so its instructions take the same arms.
+ let shadow_reps = p.shadow_reps();
+ for it in 0..ITERATIONS {
+ let sel = r[0];
+ let shadow_pass = (0..shadow_reps).flat_map(|_| p.shadow.iter().enumerate().map(|(k, i)| (INSTR_COUNT + k, i)));
+ for (k, ins) in p.instrs.iter().enumerate().chain(shadow_pass) {
+ let d = ins.dst as usize;
+ let a = ins.src as usize;
+ match ins.op {
+ Op::Scratch => {
+ // Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word).
+ let m = scratch.as_mut().expect("a scratch op needs a scratch class");
+ for lane in 0..LANES {
+ idx[lane] = r[a][lane] & slot_mask;
+ }
+ if idx.iter().all(|&x| x == idx[0]) {
+ return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
+ }
+ for lane in 0..LANES {
+ r[d][lane] = m.rmw(&p.seed, base, lane, idx[lane], r[d][lane]);
+ lane_addrs[lane * loads + nload] = 0x8000_0000 | idx[lane];
+ }
+ nload += 1;
+ }
+ Op::Add => {
+ let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32);
+ let src = r[a];
+ for lane in 0..LANES {
+ let s = (sel[lane] >> bit) & 1;
+ let c = if s != 0 { imm2 } else { imm };
+ r[d][lane] = r[d][lane].wrapping_add(src[lane]).wrapping_add(c);
+ }
+ }
+ Op::Sub => {
+ let src = r[a];
+ for lane in 0..LANES {
+ r[d][lane] = r[d][lane].wrapping_sub(src[lane]);
+ }
+ }
+ Op::Mul => {
+ let src = r[a];
+ for lane in 0..LANES {
+ r[d][lane] = r[d][lane].wrapping_mul(src[lane]);
+ }
+ }
+ Op::MulHi => {
+ let src = r[a];
+ for lane in 0..LANES {
+ r[d][lane] = mulhi32(r[d][lane], src[lane]);
+ }
+ }
+ Op::Xor => {
+ let src = r[a];
+ for lane in 0..LANES {
+ r[d][lane] ^= src[lane];
+ }
+ }
+ Op::Or => {
+ let src = r[a];
+ for lane in 0..LANES {
+ r[d][lane] |= src[lane];
+ }
+ }
+ Op::Rotl => {
+ let n = ins.rot;
+ for lane in 0..LANES {
+ r[d][lane] = r[d][lane].rotate_left(n);
+ }
+ }
+ Op::Rotr => {
+ let src = r[a];
+ for lane in 0..LANES {
+ r[d][lane] = r[d][lane].rotate_right(src[lane] & 31);
+ }
+ }
+ Op::Mad => {
+ let src = r[a];
+ let src2 = r[ins.src2 as usize];
+ for lane in 0..LANES {
+ r[d][lane] = src[lane].wrapping_mul(src2[lane]).wrapping_add(r[d][lane]);
+ }
+ }
+ Op::Shfl => {
+ let src = r[a];
+ let m = ins.mask as usize;
+ for lane in 0..LANES {
+ r[d][lane] ^= src[lane ^ m];
+ }
+ }
+ Op::Load => {
+ // Read-width experiment: a load of `width` words reads from the aligned address and folds every
+ // word (verify::fold_words); width 1 is the lottery hash's xor of one word.
+ let width = ins.width as usize;
+ let align = !(ins.width as u32 - 1);
+ let site = nload % (loads / ITERATIONS);
+ for lane in 0..LANES {
+ let v = r[a][lane];
+ acc.sat_source[site] += (v == 0 || v == u32::MAX) as u32;
+ if let Some(src) = acc.sources.as_mut() {
+ src[site].push(v);
+ }
+ idx[lane] = load_index(era.as_ref(), ins, v, mask, ACCEPT_DATASET_LOG2) & align;
+ if let Some(ix) = acc.indices.as_mut() {
+ ix[site].push(idx[lane]);
+ }
+ }
+ if idx.iter().all(|&x| x == idx[0]) {
+ return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
+ }
+ for lane in 0..LANES {
+ if width == 1 {
+ r[d][lane] ^= dataset_elem(idx[lane], d0, d1);
+ } else {
+ let mut w = [0u32; 16];
+ for j in 0..width {
+ w[j] = dataset_elem(idx[lane] + j as u32, d0, d1);
+ }
+ r[d][lane] = fold_words(r[d][lane], &w[..width]);
+ }
+ lane_addrs[lane * loads + nload] = idx[lane];
+ }
+ nload += 1;
+ }
+ Op::Hot => {
+ // Hot table: the stand-in is dataset_elem keyed by seed words 2 and 3; the address is tagged with
+ // bit 30 so a hot word and a dataset word at one index count as two addresses.
+ for lane in 0..LANES {
+ idx[lane] = hot_index(r[a][lane], hot_words);
+ }
+ if idx.iter().all(|&x| x == idx[0]) {
+ return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
+ }
+ for lane in 0..LANES {
+ r[d][lane] ^= dataset_elem(idx[lane], h0, h1);
+ lane_addrs[lane * loads + nload] = 0x4000_0000 | idx[lane];
+ }
+ nload += 1;
+ }
+ Op::WLoad => {
+ let b = (r[a][0] & mask) & !31;
+ for lane in 0..LANES {
+ idx[lane] = b + lane as u32;
+ r[d][lane] ^= dataset_elem(idx[lane], d0, d1);
+ lane_addrs[lane * loads + nload] = idx[lane];
+ }
+ nload += 1;
+ }
+ }
+ }
+ }
+ for i in 0..8 {
+ for lane in 0..LANES {
+ let v = r[i][lane];
+ acc.and_acc[i] &= v;
+ acc.or_acc[i] |= v;
+ acc.saturated += (v == 0 || v == u32::MAX) as u32;
+ }
+ }
+ for lane in 0..LANES {
+ let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21);
+ let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
+ let h = ((hi as u64) << 32) | lo as u64;
+ for j in 0..64 {
+ acc.bit_ones[j] += ((h >> j) & 1) as u32;
+ }
+ let sl = &mut lane_addrs[lane * loads..(lane + 1) * loads];
+ sl.sort_unstable();
+ let mut distinct = 0u64;
+ for k in 0..loads {
+ // scratch slots carry bit 31 (variant 5) and are not dataset addresses
+ if sl[k] & 0x8000_0000 == 0 && (k == 0 || sl[k] != sl[k - 1]) {
+ distinct += 1;
+ }
+ }
+ acc.distinct_sum += distinct;
+ }
+ Ok(())
+}
+
+/// Part (c).
+pub fn check_dynamic(p: &Program) -> Result {
+ let loads = p.loads_per_hash();
+ let v4 = is_class_v4_shape(&p.class);
+ let sites = loads / ITERATIONS;
+ let mut acc = Acc {
+ sources: None,
+ indices: None,
+ sat_source: vec![0; sites],
+ and_acc: [u32::MAX; 8],
+ or_acc: [0; 8],
+ saturated: 0,
+ bit_ones: [0; 64],
+ distinct_sum: 0,
+ };
+ let mut lane_addrs = vec![0u32; LANES * loads];
+ for (unit, &base) in accept_base_nonces(&p.seed).iter().enumerate() {
+ run_unit(p, unit, base, &mut acc, &mut lane_addrs)?;
+ }
+ for reg in 0..8 {
+ let bits = (acc.and_acc[reg] | !acc.or_acc[reg]).count_ones();
+ if bits != 0 {
+ return Err(Reject::ConstantBit { reg: reg as u8, bits: bits as u8 });
+ }
+ }
+ if acc.saturated >= MAX_SATURATED {
+ return Err(Reject::Saturated { count: acc.saturated });
+ }
+ // (c'), class v4 sub-version 2 (AP-F8-1, 7 October 2026): a load whose source is saturated reads one fixed word,
+ // whatever delivered the saturation (an or-written value, a rotate of one, a load after a saturated load); the
+ // source rule of the draw removes the writers it can see and this count catches every delivery. Keyed on the
+ // class v4 shape as the draw's rule is, so v2 and v3 verdicts do not move.
+ if v4 {
+ if let Some((site, &count)) = acc.sat_source.iter().enumerate().find(|(_, &c)| c >= MAX_SATURATED) {
+ return Err(Reject::SaturatedSource { site: site as u8, count });
+ }
+ // (c''), the ratio on the candidate that passed everything else (the draw's last and dearest test)
+ check_distinct_indices_v4(p)?;
+ }
+ let half = (ACCEPT_HASHES / 2) as u32;
+ let mut bias_max = 0u32;
+ for (bit, &ones) in acc.bit_ones.iter().enumerate() {
+ let d = ones.abs_diff(half);
+ if d > BIAS_TOLERANCE {
+ return Err(Reject::OutputBias { bit: bit as u8, ones });
+ }
+ bias_max = bias_max.max(d);
+ }
+ if acc.distinct_sum <= min_distinct_sum(loads - p.scratch_ops_per_hash()) {
+ return Err(Reject::DistinctAddresses { sum: acc.distinct_sum });
+ }
+ Ok(AcceptReport { distinct_sum: acc.distinct_sum, saturated: acc.saturated, bias_max })
+}
+
+/// The whole rule: (a), (b), then (c).
+pub fn check(p: &Program) -> Result {
+ check_static(p)?;
+ check_dynamic(p)
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+ use crate::generator::{candidate, candidate_class, generate, generate_class, GeneratorConfig, generate_v1, LoadClass};
+ use crate::verify::{DatasetMode, DatasetSource};
+
+ #[test]
+ fn distinct_bound_scales_with_the_load_count() {
+ assert_eq!(min_distinct_sum(128), MIN_DISTINCT_SUM);
+ assert_eq!(min_distinct_sum(32), 61_440);
+ }
+
+ /// The read-width classes pass the rule at about the version 2 rate, and the instrumented interpreter agrees
+ /// with `verify.rs` on every class (the fold is shared, the addresses are aligned the same way).
+ #[test]
+ fn classes_pass_and_match_verify() {
+ for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2k32", "scr8k128"] {
+ let c = LoadClass::parse(name).unwrap();
+ let p = generate_class("igneum-genesis", c);
+ assert!(check(&p).is_ok(), "{name}");
+ let mut rejected = 0;
+ for i in 0..60u32 {
+ let s = format!("igneum-rw-accept/{i}");
+ let q = candidate_class(&s, s.as_bytes(), 0, c);
+ if check(&q).is_err() {
+ rejected += 1;
+ }
+ }
+ assert!(rejected < 15, "{name}: {rejected} of 60 rejected");
+ let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2);
+ let bases = accept_base_nonces(&p.seed);
+ let loads = p.loads_per_hash();
+ let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
+ let mut la = vec![0u32; LANES * loads];
+ let mut ones = [0u32; 64];
+ for (u, &b) in bases.iter().enumerate() {
+ run_unit(&p, u, b, &mut acc, &mut la).unwrap();
+ for h in crate::verify::hash_warp(&p, b, &ds) {
+ for j in 0..64 {
+ ones[j] += ((h >> j) & 1) as u32;
+ }
+ }
+ }
+ assert_eq!(acc.bit_ones, ones, "{name}: bit counts match the reference interpreter");
+ }
+ }
+
+ /// AP-F8-3 (7 October 2026): the acceptance's execution and `verify.rs` agree on a class v4 program WITH its
+ /// shadow block (the output bit counts over the 64 units on the closed-form dataset, the same sel per iteration),
+ /// so the two paths cannot diverge again: until sub-version 3 the acceptance ran the base program only and judged
+ /// a program the chain never hashes. The devnet epoch-0 seed and the six test eras, 8 x 256 x 27 shadow
+ /// instructions per hash each; the same program with its shadow stripped gives other counts.
+ #[test]
+ fn acceptance_executes_the_shadow_block_as_the_verifier_does() {
+ use crate::generator::{generate_era, EraParams, V3_ALLOWED, V4_CLASS};
+ let hx = |h: &str| -> Vec { (0..h.len()).step_by(2).map(|i| u8::from_str_radix(&h[i..i + 2], 16).unwrap()).collect() };
+ let g = hx("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07");
+ let mut eras = vec![g.clone()];
+ for n in 0..6 {
+ eras.push(EraParams::test_era_bytes(&format!("igneum-era-test/{n}")).to_vec());
+ }
+ for era in &eras {
+ let p = generate_era("igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07", &g, V4_CLASS, era, &V3_ALLOWED);
+ assert_eq!(p.shadow.len(), 256);
+ assert_eq!(p.shadow_reps(), 27);
+ let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2);
+ let bases = accept_base_nonces(&p.seed);
+ let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
+ let mut la = vec![0u32; LANES * p.loads_per_hash()];
+ let mut ones = [0u32; 64];
+ for (u, &b) in bases.iter().enumerate() {
+ run_unit(&p, u, b, &mut acc, &mut la).unwrap();
+ for h in crate::verify::hash_warp(&p, b, &ds) {
+ for j in 0..64 {
+ ones[j] += ((h >> j) & 1) as u32;
+ }
+ }
+ }
+ assert_eq!(acc.bit_ones, ones, "the acceptance's execution of a class v4 program (shadow block included) matches the verifier's hashes");
+ // and the same program with its shadow stripped hashes differently: the shadow is executed, not skipped
+ let mut bare = p.clone();
+ bare.shadow.clear();
+ let mut acc2 = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
+ for (u, &b) in bases.iter().enumerate() {
+ let _ = run_unit(&bare, u, b, &mut acc2, &mut la);
+ }
+ assert_ne!(acc.bit_ones, acc2.bit_ones, "the shadow block changes the acceptance's execution");
+ }
+ }
+
+ /// The instrumented interpreter agrees with `verify.rs` on the closed-form dataset keyed by the seed words.
+ #[test]
+ fn instrumented_interpreter_matches_verify() {
+ for i in 0..20u32 {
+ let s = format!("igneum-accept-test/{i}");
+ let p = candidate(&s, s.as_bytes(), 0);
+ let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2);
+ let bases = accept_base_nonces(&p.seed);
+ let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
+ let mut la = vec![0u32; LANES * p.loads_per_hash()];
+ let mut ones = [0u32; 64];
+ let mut any = false;
+ for (u, &b) in bases.iter().enumerate() {
+ if run_unit(&p, u, b, &mut acc, &mut la).is_err() {
+ continue;
+ }
+ any = true;
+ let w = crate::verify::hash_warp(&p, b, &ds);
+ for h in w {
+ for j in 0..64 {
+ ones[j] += ((h >> j) & 1) as u32;
+ }
+ }
+ }
+ if any {
+ assert_eq!(acc.bit_ones, ones, "bit counts of {s} match the reference interpreter's hashes");
+ }
+ }
+ }
+
+ /// Hot-table experiment: the hot classes pass the rule at about the version 2 rate, and the hot addresses are
+ /// uniform over the table (16 buckets of the index over 64 units x 32 lanes x 32 hot loads).
+ #[test]
+ fn hot_classes_pass_and_hot_loads_are_uniform() {
+ for name in ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "scr4k32+hot64k4", "hot32k4a", "hot64k4a", "hot96k4a"] {
+ let c = LoadClass::parse(name).unwrap();
+ let p = generate_class("igneum-genesis", c);
+ assert!(check(&p).is_ok(), "{name}");
+ let mut rejected = 0;
+ for i in 0..60u32 {
+ let s = format!("igneum-hot-accept/{i}");
+ let q = candidate_class(&s, s.as_bytes(), 0, c);
+ if check(&q).is_err() {
+ rejected += 1;
+ }
+ }
+ assert!(rejected < 15, "{name}: {rejected} of 60 rejected");
+ }
+ let p = generate_class("igneum-genesis", LoadClass::hot(96, 4));
+ let words = p.hot_words();
+ let loads = p.loads_per_hash();
+ let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
+ let mut la = vec![0u32; LANES * loads];
+ let mut buckets = [0u64; 16];
+ let mut hot_count = 0u64;
+ for (u, &b) in accept_base_nonces(&p.seed).iter().enumerate() {
+ run_unit(&p, u, b, &mut acc, &mut la).unwrap();
+ for &a in &la {
+ if a & 0xC000_0000 == 0x4000_0000 {
+ let idx = a & 0x3FFF_FFFF;
+ assert!(idx < words);
+ buckets[(idx as u64 * 16 / words as u64) as usize] += 1;
+ hot_count += 1;
+ }
+ }
+ }
+ assert_eq!(hot_count, 64 * 32 * 32, "32 hot loads per hash over 2,048 hashes");
+ let mean = hot_count as f64 / 16.0;
+ for (i, &b) in buckets.iter().enumerate() {
+ assert!((b as f64 - mean).abs() < 0.15 * mean, "bucket {i}: {b} against a mean of {mean}");
+ }
+ // the dataset distinct count still holds for the dataset loads alone
+ let r = check(&p).unwrap();
+ assert!(r.distinct_mean() > 120.0);
+ }
+
+ #[test]
+ fn base_nonces_are_aligned_and_seed_dependent() {
+ let a = accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 8]);
+ let b = accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 9]);
+ assert!(a.iter().all(|x| x & 31 == 0));
+ assert_ne!(a, b);
+ assert_eq!(a, accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 8]));
+ }
+
+ #[test]
+ fn stale_load_detection_is_cyclic() {
+ let mut p = candidate("igneum-genesis", b"igneum-genesis", 0);
+ assert!(check_stale_loads(&p.instrs).is_ok(), "an accepted candidate has no stale load");
+ // Make the last instruction a load from r3 and the first a load from r3 with no write between (wrap).
+ let (first, last) = (0usize, INSTR_COUNT - 1);
+ p.instrs[last].op = Op::Load;
+ p.instrs[last].src = 3;
+ p.instrs[last].dst = 4;
+ p.instrs[first].op = Op::Load;
+ p.instrs[first].src = 3;
+ p.instrs[first].dst = 5;
+ assert_eq!(check_stale_loads(&p.instrs), Err(Reject::StaleLoadSource { instr: 0, reg: 3 }));
+ }
+
+ #[test]
+ fn injecting_write_detection() {
+ let mut p = candidate("igneum-genesis", b"igneum-genesis", 0);
+ for ins in p.instrs.iter_mut() {
+ if ins.dst == 6 && ins.op.injects() {
+ ins.op = Op::Rotl;
+ }
+ }
+ assert_eq!(check_injecting_writes(&p.instrs), Err(Reject::NoInjectingWrite { reg: 6 }));
+ }
+
+ /// The census's measured rates: about 5 percent of candidates rejected, 128 distinct loads for the rest.
+ #[test]
+ fn rejection_rate_and_distinct_loads_on_a_sample() {
+ let mut rejected = 0;
+ let mut dsum = 0.0;
+ let mut accepted = 0;
+ for i in 0..400u32 {
+ let s = format!("igneum-census-2026-10-03/{i}");
+ match check(&candidate(&s, s.as_bytes(), 0)) {
+ Ok(r) => {
+ accepted += 1;
+ dsum += r.distinct_mean();
+ assert!(r.distinct_mean() > 120.0);
+ }
+ Err(_) => rejected += 1,
+ }
+ }
+ assert!(rejected < 50, "{rejected} of 400 rejected");
+ assert!(dsum / accepted as f64 > 127.0, "mean distinct {}", dsum / accepted as f64);
+ }
+
+ /// The retired generator fails the rule on nearly every program (the census: 95 percent).
+ #[test]
+ fn v1_programs_are_mostly_rejected() {
+ let mut rejected = 0;
+ for i in 0..100u32 {
+ let s = format!("igneum-census-2026-10-03/{i}");
+ if check(&generate_v1(&s, &GeneratorConfig::default())).is_err() {
+ rejected += 1;
+ }
+ }
+ assert!(rejected > 80, "{rejected} of 100 rejected");
+ }
+
+ #[test]
+ fn generated_programs_pass() {
+ for s in ["igneum-genesis", "igneum-hourly", "igneum-second-seed"] {
+ assert!(check(&generate(s)).is_ok());
+ }
+ }
+}
diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/bind.rs b/tools/attack/adv-accept-v5/igneum-pow/src/bind.rs
new file mode 100644
index 000000000..94edea2bd
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/src/bind.rs
@@ -0,0 +1,234 @@
+//! Header binding: the init words of the lottery hash commit to the block being mined.
+//!
+//! `docs/spec/01-lottery-hash.md` section 1.6 (open item O-1.9) proposes the rule implemented here:
+//!
+//! * The header nonce is 64 bits. Its low 32 bits are the lane nonce `n` (the per-thread nonce of the kernel).
+//! * Its high 32 bits and the 256-bit pre-PoW header hash `H` form the init words:
+//! `I = seed_words_from_bytes("igneum-block/" || H || nonce_hi_le32)`.
+//! * `I` is a kernel argument, not a compile-time constant. The program (from the epoch seed) is compiled once per
+//! epoch; `I` changes per block template.
+//!
+//! Choices the spec leaves open, fixed here (3 October 2026):
+//!
+//! | Choice | Rule | Why |
+//! |---|---|---|
+//! | `H` | the header hash with the nonce field set to zero and every other field as mined (rusty-kaspa `hash_override_nonce_time(header, 0, header.timestamp)`) | Kaspa's pre-PoW hash also zeroes the timestamp and absorbs it later in cSHAKE. The lane hash has no later step, so the timestamp must be inside `H` or a miner could reuse one nonce for many timestamps |
+//! | 256-bit pow value | lane hash in the top 64 bits (little-endian bytes 24..32), low 192 bits zero | `pow <= target256` is then exactly `lane <= target256 >> 192` (section 1.10 candidate), so a GPU worker and the node compare the same 64-bit numbers; the block level is `leading_zeros(lane)` shifted |
+//! | Day seed bytes (O-1.10 interim) | `"igneum-day/" || day_le64` with `day = header.timestamp_ms / 86,400,000` | The spec's proposal ties the day key to the first epoch seed of the day, which needs the VDF schedule. The interim rule keeps one 256 MiB cache per calendar day and needs no chain walk |
+//! | Epoch seed bytes | the 32 bytes of the epoch block hash (devnet v0: the last selected-chain block below the epoch's start DAA score, genesis for epoch 0) | Section 1.12: `S_e = seed_words_from_bytes(program_seed_e)`; the VDF output replaces the block hash later without touching this crate |
+//!
+//! The packs' vectors (init words equal to the program seed) stay the conformance vectors for the generator,
+//! interpreter and dataset. The bound vectors are in `README.md` and in the tests below (re-cut for generator
+//! version 2 on 4 October 2026).
+
+use crate::generator::LANES;
+use crate::seed::seed_words_from_bytes;
+use crate::verify::{interpret_warp_init, Epoch, WarpResult};
+
+/// Domain tag of the init words.
+pub const BLOCK_TAG: &[u8] = b"igneum-block/";
+/// Domain tag of the interim day seed.
+pub const DAY_TAG: &[u8] = b"igneum-day/";
+/// Milliseconds per day, the clock of the interim day seed.
+pub const DAY_MS: u64 = 86_400_000;
+
+/// The lane nonce: low 32 bits of the header nonce.
+#[inline]
+pub fn lane_nonce(nonce: u64) -> u32 {
+ nonce as u32
+}
+
+/// The high 32 bits of the header nonce (the extra nonce that enters the init words).
+#[inline]
+pub fn nonce_hi(nonce: u64) -> u32 {
+ (nonce >> 32) as u32
+}
+
+/// `"igneum-block/" || H || nonce_hi_le32`, the bytes the init words are derived from.
+pub fn block_init_bytes(header_prehash: &[u8; 32], nonce: u64) -> [u8; 49] {
+ let mut b = [0u8; 49];
+ b[..13].copy_from_slice(BLOCK_TAG);
+ b[13..45].copy_from_slice(header_prehash);
+ b[45..49].copy_from_slice(&nonce_hi(nonce).to_le_bytes());
+ b
+}
+
+/// The init words `I` for a header and a 64-bit nonce (only the high 32 bits of the nonce matter).
+pub fn block_init_words(header_prehash: &[u8; 32], nonce: u64) -> [u32; 8] {
+ seed_words_from_bytes(&block_init_bytes(header_prehash, nonce))
+}
+
+/// Interim day seed bytes: `"igneum-day/" || day_le64`.
+pub fn day_bytes(day_index: u64) -> [u8; 19] {
+ let mut b = [0u8; 19];
+ b[..11].copy_from_slice(DAY_TAG);
+ b[11..19].copy_from_slice(&day_index.to_le_bytes());
+ b
+}
+
+/// Day index of a header timestamp in milliseconds.
+#[inline]
+pub fn day_index(timestamp_ms: u64) -> u64 {
+ timestamp_ms / DAY_MS
+}
+
+/// The 256-bit pow value as little-endian bytes: the lane hash in bytes 24..32, zero elsewhere.
+pub fn pow256_from_lane(lane: u64) -> [u8; 32] {
+ let mut b = [0u8; 32];
+ b[24..32].copy_from_slice(&lane.to_le_bytes());
+ b
+}
+
+/// The 64-bit target from a little-endian 256-bit target: its top 64 bits.
+pub fn target64_from_le256(target: &[u8; 32]) -> u64 {
+ u64::from_le_bytes(target[24..32].try_into().unwrap())
+}
+
+/// Lower-case hex of bytes.
+pub fn hex(bytes: &[u8]) -> String {
+ bytes.iter().map(|b| format!("{b:02x}")).collect()
+}
+
+/// Bytes from hex (either case). `None` on odd length or a bad digit.
+pub fn unhex(s: &str) -> Option> {
+ if s.len() % 2 != 0 {
+ return None;
+ }
+ (0..s.len()).step_by(2).map(|i| u8::from_str_radix(&s[i..i + 2], 16).ok()).collect()
+}
+
+impl Epoch {
+ /// The 32 bound hashes of the aligned warp that contains `nonce`: lane `l` is the hash of
+ /// `(nonce_hi << 32) | ((lane_nonce & !31) + l)`.
+ pub fn hash_warp_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> [u64; LANES] {
+ self.interpret_warp_bound(header_prehash, nonce).hashes
+ }
+
+ pub fn interpret_warp_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> WarpResult {
+ let init = block_init_words(header_prehash, nonce);
+ interpret_warp_init(&self.program, &init, lane_nonce(nonce) & !31, &self.dataset)
+ }
+
+ /// The 32 bound hashes for already-derived init words (what a GPU worker computes per dispatch).
+ pub fn hash_warp_init(&self, init: &[u32; 8], base_lane_nonce: u32) -> [u64; LANES] {
+ interpret_warp_init(&self.program, init, base_lane_nonce, &self.dataset).hashes
+ }
+
+ /// The bound 64-bit lane hash of one header nonce.
+ pub fn hash_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> u64 {
+ self.hash_warp_bound(header_prehash, nonce)[(lane_nonce(nonce) & 31) as usize]
+ }
+
+ /// The bound 256-bit pow value (little-endian): the lane hash in the top 64 bits.
+ pub fn pow_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> [u8; 32] {
+ pow256_from_lane(self.hash_bound(header_prehash, nonce))
+ }
+
+ /// `hash_bound(H, nonce) <= target64`.
+ pub fn verify_block_bound(&self, header_prehash: &[u8; 32], nonce: u64, target64: u64) -> bool {
+ self.hash_bound(header_prehash, nonce) <= target64
+ }
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+ use crate::verify::DatasetMode;
+ use std::sync::OnceLock;
+
+ fn epoch() -> &'static Epoch {
+ static E: OnceLock = OnceLock::new();
+ E.get_or_init(|| Epoch::memory_hard("igneum-genesis", "2026-10-03"))
+ }
+
+ fn prehash_a() -> [u8; 32] {
+ [0u8; 32]
+ }
+ fn prehash_b() -> [u8; 32] {
+ let mut h = [0u8; 32];
+ for (i, b) in h.iter_mut().enumerate() {
+ *b = i as u8;
+ }
+ h
+ }
+
+ #[test]
+ fn init_bytes_layout() {
+ let b = block_init_bytes(&prehash_b(), 0x0000_0102_0000_0007);
+ assert_eq!(&b[..13], b"igneum-block/");
+ assert_eq!(&b[13..45], &prehash_b());
+ assert_eq!(&b[45..49], &[0x02, 0x01, 0x00, 0x00]);
+ assert_eq!(
+ block_init_words(&prehash_b(), 0x0000_0102_0000_0007),
+ block_init_words(&prehash_b(), 0x0000_0102_ffff_ffff)
+ );
+ assert_ne!(block_init_words(&prehash_b(), 0), block_init_words(&prehash_b(), 1 << 32));
+ }
+
+ #[test]
+ fn day_bytes_layout() {
+ let b = day_bytes(20_729);
+ assert_eq!(&b[..11], b"igneum-day/");
+ assert_eq!(&b[11..], &20_729u64.to_le_bytes());
+ assert_eq!(day_index(0x1a0ff0f7c00), 20_729);
+ }
+
+ #[test]
+ fn pow256_and_target64() {
+ let p = pow256_from_lane(0x0123_4567_89ab_cdef);
+ assert_eq!(&p[..24], &[0u8; 24]);
+ assert_eq!(target64_from_le256(&p), 0x0123_4567_89ab_cdef);
+ assert_eq!(hex(&p[24..]), "efcdab8967452301");
+ assert_eq!(unhex("efcdab8967452301").unwrap(), p[24..].to_vec());
+ assert!(unhex("abc").is_none());
+ }
+
+ /// The bound form is a different hash from the unbound one, depends on H and on the high nonce bits, and the
+ /// single-nonce form agrees with the warp form.
+ #[test]
+ fn bound_hash_properties() {
+ let e = epoch();
+ let a = prehash_a();
+ let b = prehash_b();
+ assert_ne!(e.hash_bound(&a, 0), e.hash(0), "bound differs from the pack vector");
+ assert_ne!(e.hash_bound(&a, 0), e.hash_bound(&b, 0), "H enters the hash");
+ assert_ne!(e.hash_bound(&a, 0), e.hash_bound(&a, 1 << 32), "nonce_hi enters the hash");
+ let w = e.hash_warp_bound(&a, (1 << 32) | 37);
+ assert_eq!(w[5], e.hash_bound(&a, (1 << 32) | 37));
+ assert_eq!(w[0], e.hash_bound(&a, 1 << 32 | 32));
+ let init = block_init_words(&a, 1 << 32);
+ assert_eq!(e.hash_warp_init(&init, 32), w);
+ let t = e.hash_bound(&a, 7);
+ assert!(e.verify_block_bound(&a, 7, t));
+ assert!(!e.verify_block_bound(&a, 7, t - 1));
+ assert_eq!(e.pow_bound(&a, 7), pow256_from_lane(t));
+ }
+
+ /// The 8 bound vectors printed in README.md (seed igneum-genesis, day 2026-10-03, memory-hard, 2^28 words,
+ /// generator v2 since 4 October 2026).
+ #[test]
+ fn bound_vectors() {
+ let e = epoch();
+ let a = prehash_a();
+ let b = prehash_b();
+ let cases: [(&[u8; 32], u64, u64); 8] = [
+ (&a, 0, 0x746c567b090acf6a),
+ (&a, 1, 0x45a619f860880c73),
+ (&a, 31, 0x9aa495e43dedbfe6),
+ (&a, 4096, 0x2e6ffd7624d3cba2),
+ (&a, 1 << 32, 0x38a5cea1fb01431a),
+ (&b, 0, 0x2a79c5e4797bf6aa),
+ (&b, (1 << 32) | 5, 0xa243e0c61aa1b82e),
+ (&b, u64::MAX, 0x9c7bbfbd064fe1a4),
+ ];
+ for (h, nonce, want) in cases {
+ assert_eq!(e.hash_bound(h, nonce), want, "H {} nonce {nonce}", hex(h));
+ }
+ }
+
+ #[test]
+ fn closed_form_bound_also_works() {
+ let e = Epoch::new("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 28);
+ assert_ne!(e.hash_bound(&prehash_a(), 0), e.hash(0));
+ }
+}
diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/blake2b.rs b/tools/attack/adv-accept-v5/igneum-pow/src/blake2b.rs
new file mode 100644
index 000000000..4b04cc622
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/src/blake2b.rs
@@ -0,0 +1,152 @@
+//! BLAKE2b (RFC 7693), the chain's own hash family (spec 01 section 0.6), written out here so the crate keeps its
+//! rule of no dependency outside the standard library. Used by class v5's state leaves (`crate::state`):
+//! `blake2b_512` for a leaf digest, `blake2b_256` for the sample order. Unkeyed, no salt, no personalisation.
+//! Checked against the RFC's "abc" vector and the empty-input vector in the tests.
+
+const IV: [u64; 8] = [
+ 0x6a09e667f3bcc908,
+ 0xbb67ae8584caa73b,
+ 0x3c6ef372fe94f82b,
+ 0xa54ff53a5f1d36f1,
+ 0x510e527fade682d1,
+ 0x9b05688c2b3e6c1f,
+ 0x1f83d9abfb41bd6b,
+ 0x5be0cd19137e2179,
+];
+
+const SIGMA: [[usize; 16]; 12] = [
+ [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15],
+ [14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3],
+ [11, 8, 12, 0, 5, 2, 15, 13, 10, 14, 3, 6, 7, 1, 9, 4],
+ [7, 9, 3, 1, 13, 12, 11, 14, 2, 6, 5, 10, 4, 0, 15, 8],
+ [9, 0, 5, 7, 2, 4, 10, 15, 14, 1, 11, 12, 6, 8, 3, 13],
+ [2, 12, 6, 10, 0, 11, 8, 3, 4, 13, 7, 5, 15, 14, 1, 9],
+ [12, 5, 1, 15, 14, 13, 4, 10, 0, 7, 6, 3, 9, 2, 8, 11],
+ [13, 11, 7, 14, 12, 1, 3, 9, 5, 0, 15, 4, 8, 6, 2, 10],
+ [6, 15, 14, 9, 11, 3, 0, 8, 12, 2, 13, 7, 1, 4, 10, 5],
+ [10, 2, 8, 4, 7, 6, 1, 5, 15, 11, 9, 14, 3, 12, 13, 0],
+ [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15],
+ [14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3],
+];
+
+#[inline(always)]
+fn g(v: &mut [u64; 16], a: usize, b: usize, c: usize, d: usize, x: u64, y: u64) {
+ v[a] = v[a].wrapping_add(v[b]).wrapping_add(x);
+ v[d] = (v[d] ^ v[a]).rotate_right(32);
+ v[c] = v[c].wrapping_add(v[d]);
+ v[b] = (v[b] ^ v[c]).rotate_right(24);
+ v[a] = v[a].wrapping_add(v[b]).wrapping_add(y);
+ v[d] = (v[d] ^ v[a]).rotate_right(16);
+ v[c] = v[c].wrapping_add(v[d]);
+ v[b] = (v[b] ^ v[c]).rotate_right(63);
+}
+
+fn compress(h: &mut [u64; 8], block: &[u8; 128], t: u128, last: bool) {
+ let mut m = [0u64; 16];
+ for (i, w) in m.iter_mut().enumerate() {
+ *w = u64::from_le_bytes(block[i * 8..i * 8 + 8].try_into().unwrap());
+ }
+ let mut v = [0u64; 16];
+ v[..8].copy_from_slice(h);
+ v[8..].copy_from_slice(&IV);
+ v[12] ^= t as u64;
+ v[13] ^= (t >> 64) as u64;
+ if last {
+ v[14] = !v[14];
+ }
+ for s in SIGMA.iter() {
+ g(&mut v, 0, 4, 8, 12, m[s[0]], m[s[1]]);
+ g(&mut v, 1, 5, 9, 13, m[s[2]], m[s[3]]);
+ g(&mut v, 2, 6, 10, 14, m[s[4]], m[s[5]]);
+ g(&mut v, 3, 7, 11, 15, m[s[6]], m[s[7]]);
+ g(&mut v, 0, 5, 10, 15, m[s[8]], m[s[9]]);
+ g(&mut v, 1, 6, 11, 12, m[s[10]], m[s[11]]);
+ g(&mut v, 2, 7, 8, 13, m[s[12]], m[s[13]]);
+ g(&mut v, 3, 4, 9, 14, m[s[14]], m[s[15]]);
+ }
+ for i in 0..8 {
+ h[i] ^= v[i] ^ v[i + 8];
+ }
+}
+
+/// Unkeyed BLAKE2b of `data` with an output of `out_len` bytes (1..=64), written into `out[..out_len]`.
+pub fn blake2b(out: &mut [u8], out_len: usize, data: &[u8]) {
+ assert!((1..=64).contains(&out_len) && out.len() >= out_len);
+ let mut h = IV;
+ h[0] ^= 0x0101_0000 ^ out_len as u64;
+ let mut t: u128 = 0;
+ let n = data.len();
+ // every full block but the last; the last block (possibly empty) is compressed with the final flag
+ let full = if n == 0 { 0 } else { (n - 1) / 128 };
+ for i in 0..full {
+ let block: &[u8; 128] = data[i * 128..i * 128 + 128].try_into().unwrap();
+ t += 128;
+ compress(&mut h, block, t, false);
+ }
+ let mut last = [0u8; 128];
+ let rest = &data[full * 128..];
+ last[..rest.len()].copy_from_slice(rest);
+ t += rest.len() as u128;
+ compress(&mut h, &last, t, true);
+ let mut bytes = [0u8; 64];
+ for (i, w) in h.iter().enumerate() {
+ bytes[i * 8..i * 8 + 8].copy_from_slice(&w.to_le_bytes());
+ }
+ out[..out_len].copy_from_slice(&bytes[..out_len]);
+}
+
+/// BLAKE2b-512 of the concatenation of `parts`.
+pub fn blake2b_512(parts: &[&[u8]]) -> [u8; 64] {
+ let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum());
+ for p in parts {
+ data.extend_from_slice(p);
+ }
+ let mut out = [0u8; 64];
+ blake2b(&mut out, 64, &data);
+ out
+}
+
+/// BLAKE2b-256 of the concatenation of `parts`.
+pub fn blake2b_256(parts: &[&[u8]]) -> [u8; 32] {
+ let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum());
+ for p in parts {
+ data.extend_from_slice(p);
+ }
+ let mut out = [0u8; 32];
+ blake2b(&mut out, 32, &data);
+ out
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+
+ fn hex(b: &[u8]) -> String {
+ b.iter().map(|x| format!("{x:02x}")).collect()
+ }
+
+ /// RFC 7693 appendix A ("abc"), the empty input, and a two-block input against the reference implementation's
+ /// known values (the three-block "The quick brown fox" vector of the BLAKE2 test suite).
+ #[test]
+ fn rfc_7693_vectors() {
+ assert_eq!(
+ hex(&blake2b_512(&[b"abc"])),
+ "ba80a53f981c4d0d6a2797b69f12f6e94c212f14685ac4b74b12bb6fdbffa2d17d87c5392aab792dc252d5de4533cc9518d38aa8dbf1925ab92386edd4009923"
+ );
+ assert_eq!(
+ hex(&blake2b_512(&[b""])),
+ "786a02f742015903c6c6fd852552d272912f4740e15847618a86e217f71f5419d25e1031afee585313896444934eb04b903a685b1448b755d56f701afe9be2ce"
+ );
+ assert_eq!(hex(&blake2b_256(&[b"abc"])), "bddd813c634239723171ef3fee98579b94964e3bb1cb3e427262c8c068d52319");
+ assert_eq!(hex(&blake2b_256(&[b""])), "0e5751c026e543b2e8ab2eb06099daa1d1e5df47778f7787faab45cdf12fe3a8");
+ // a 128-byte input is exactly one full block compressed as the last; 129 bytes takes two
+ let one = [0x61u8; 128];
+ let two = [0x61u8; 129];
+ assert_ne!(blake2b_512(&[&one]), blake2b_512(&[&two]));
+ assert_eq!(blake2b_512(&[&one[..64], &one[64..]]), blake2b_512(&[&one]), "parts concatenate");
+ assert_eq!(
+ hex(&blake2b_512(&[b"The quick brown fox jumps over the lazy dog"])),
+ "a8add4bdddfd93e4877d2746e62817b116364a1fa7bc148d95090bc7333b3673f82401cf7aa2e4cb1ecd90296e3f14cb5413f8ed77be73045b13914cdcd6a918"
+ );
+ }
+}
diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/derive.rs b/tools/attack/adv-accept-v5/igneum-pow/src/derive.rs
new file mode 100644
index 000000000..7e8c5021f
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/src/derive.rs
@@ -0,0 +1,978 @@
+//! The per-day item-derivation program (Counter ASIC 3.0 item 2, `docs/plans/counter-asic-3-derivation.md`):
+//! RandomX's SuperscalarHash idea (`vendor/RandomX/src/superscalar.cpp`, read 6 October 2026 at commit 7607fb2)
+//! rebuilt for a 16-word item on a GPU. In place of the fixed-shape mixer `M_r` of spec 01 section 1.8.4, each of
+//! the nine mixer slots of an item (one before each of the 8 dependent cache reads, one after the last) runs a
+//! straight-line program of [`DERIVE_LEN`] instructions drawn once a day from the day key stream, from a fixed
+//! set of twelve two-register forms. The 8 dependent cache reads per item are untouched.
+//!
+//! Rules of the draw (the dependency chain of SuperscalarHash, made strict):
+//! * every instruction reads the chain register `c`, the register the previous instruction wrote (`s[0]`, the
+//! address word, at the start of each round program), and writes a register `d != c`, which becomes the chain;
+//! so no two instructions of a program can run in parallel, and no two consecutive instructions write one
+//! register (the "ror r,C1; ror r,C2" and "xor r,r2; xor r,r2" merges of SuperscalarHash's `selectDestination`
+//! cannot arise);
+//! * every form is a bijection on the 16-word state (the old `d` enters through `+=`, `-=`, `^=`, an odd multiply,
+//! or a rotation of itself), so a program loses no entropy, the property `M_r` has;
+//! * the forms are integer only, modulo 2^32, with rotations by 1..31: bit-exact on Metal, CUDA and OpenCL by the
+//! same argument as the lottery hash's families (spec 01 section 1.14); no division, no float, no branch;
+//! * four draws per instruction in a fixed order, so the stream position of every draw is fixed by the index.
+//!
+//! The acceptance test ([`DeriveProgram::check`]) rejects a degenerate draw and the next attempt is drawn from the
+//! continuation of the stream, the rule the program generator uses (spec 01 section 1.4.6).
+
+use crate::memhard::ITEM_ROUNDS;
+use crate::seed::SplitMix64;
+
+/// Registers of the item state (the item is 16 words).
+pub const DERIVE_REGS: usize = 16;
+/// Round programs per item: one before each cache read and one after the last (`ITEM_ROUNDS + 1`).
+pub const DERIVE_PROGRAMS: usize = ITEM_ROUNDS + 1;
+/// Instructions per round program for the x8-equivalent operation count (the candidate, class "dr736"): 9 x 736
+/// = 6,624 instructions per item at a mean of 1.62 GPU operations (1.52 chip operations) each, about 10,730 GPU
+/// operations, 10,070 chip operations and 1,460 multiplies per item. The x8 mixer, counted from the code
+/// (`memhard::mixer`, 16 x (xor, add, mul) + 8 quarter rounds x 12 = 144 operations as written, 128 with the
+/// `RC + rk` adds hoisted as constants, 16 multiplies; `chip-model-v3.md` section 1 prices 130 from the spec text):
+/// 72 applications = 10,368 as written, 9,216 hoisted, 1,152 multiplies. The floors below are those three.
+pub const DERIVE_LEN_X8: u32 = 736;
+/// Draws per instruction: the op roll, the destination roll, the second-source roll and the immediate.
+pub const DRAWS_PER_INSTR: u64 = 4;
+/// The floor of chip operations per item (the x8 mixer with its constants hoisted: 72 x 128).
+pub const OPS_FLOOR_X8: u64 = 9_216;
+/// The floor of GPU operations per item (the x8 mixer as written: 72 x 144).
+pub const GPU_OPS_FLOOR_X8: u64 = 10_368;
+/// The floor of multiplies per item (the x8 mixer's 72 x 16).
+pub const MULS_FLOOR_X8: u64 = 1_152;
+/// Distinct rotation amounts an item's programs must use, at least.
+pub const DISTINCT_ROTS_FLOOR: usize = 8;
+/// Attempts before the generator gives up (never reached: see [`DeriveProgram::draw`]).
+pub const MAX_ATTEMPTS: u32 = 64;
+
+/// The twelve forms. `c` is the chain register (the previous destination), `d` the destination (`d != c`), `b` a
+/// third register (`b != d`, `b != c`), `k` a rotation in 1..31, `i` a 32-bit constant (odd for `MulC`).
+#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
+#[repr(u8)]
+pub enum DOp {
+ /// `d += c`
+ Add = 0,
+ /// `d -= c`
+ Sub = 1,
+ /// `d ^= c`
+ Xor = 2,
+ /// `d *= (c OR 1)`: the multiply-lo form, odd so it is a bijection on `d`
+ Mul = 3,
+ /// `d = rotl(d, k) + c`
+ Rot = 4,
+ /// `d = rotl(d ^ c, k)`
+ XRot = 5,
+ /// `d += c + i`
+ AddC = 6,
+ /// `d ^= c ^ i`
+ XorC = 7,
+ /// `d = (d ^ c) * i`, `i` odd: the per-word form of `M_r` with the chain in place of the round constant
+ MulC = 8,
+ /// `d = d * i + c`, `i` odd
+ MulC2 = 9,
+ /// `d ^= (c AND b)`
+ AndX = 10,
+ /// `d += (c OR b)`
+ OrX = 11,
+}
+
+/// The op weights in percent, in draw order (sum 100). Fixed at genesis; only the order, the registers and the
+/// constants are drawn.
+pub const DOP_WEIGHTS: [(DOp, u64); 12] = [
+ (DOp::Add, 14),
+ (DOp::Sub, 10),
+ (DOp::Xor, 14),
+ (DOp::Mul, 10),
+ (DOp::Rot, 10),
+ (DOp::XRot, 10),
+ (DOp::AddC, 6),
+ (DOp::XorC, 6),
+ (DOp::MulC, 8),
+ (DOp::MulC2, 4),
+ (DOp::AndX, 4),
+ (DOp::OrX, 4),
+];
+
+impl DOp {
+ pub fn from_u8(v: u8) -> Option {
+ DOP_WEIGHTS.iter().map(|(o, _)| *o).find(|o| *o as u8 == v)
+ }
+ pub fn name(self) -> &'static str {
+ match self {
+ DOp::Add => "add",
+ DOp::Sub => "sub",
+ DOp::Xor => "xor",
+ DOp::Mul => "mul",
+ DOp::Rot => "rot",
+ DOp::XRot => "xrot",
+ DOp::AddC => "addc",
+ DOp::XorC => "xorc",
+ DOp::MulC => "mulc",
+ DOp::MulC2 => "mulc2",
+ DOp::AndX => "andx",
+ DOp::OrX => "orx",
+ }
+ }
+ /// Integer operations as a GPU executes the form (every `|`, `&`, `+`, `^`, `*`, rotate counts one).
+ pub fn gpu_ops(self) -> u64 {
+ match self {
+ DOp::Add | DOp::Sub | DOp::Xor => 1,
+ _ => 2,
+ }
+ }
+ /// Integer operations as the chip model counts them (`c OR 1` is a wire on a chip, so `Mul` is one multiply;
+ /// a constant folded into a chain value is still an add or an xor, so every other two-op form stays two).
+ pub fn chip_ops(self) -> u64 {
+ match self {
+ DOp::Add | DOp::Sub | DOp::Xor | DOp::Mul => 1,
+ _ => 2,
+ }
+ }
+ pub fn is_mul(self) -> bool {
+ matches!(self, DOp::Mul | DOp::MulC | DOp::MulC2)
+ }
+ pub fn has_rot(self) -> bool {
+ matches!(self, DOp::Rot | DOp::XRot)
+ }
+ pub fn has_third(self) -> bool {
+ matches!(self, DOp::AndX | DOp::OrX)
+ }
+ pub fn has_imm(self) -> bool {
+ matches!(self, DOp::AddC | DOp::XorC | DOp::MulC | DOp::MulC2)
+ }
+ /// The op of a roll in 0..99.
+ pub fn for_roll(roll: u64) -> DOp {
+ let mut acc = 0u64;
+ for (op, w) in DOP_WEIGHTS {
+ acc += w;
+ if roll < acc {
+ return op;
+ }
+ }
+ DOp::OrX
+ }
+}
+
+/// One instruction. `src` is the chain register (carried so the interpreter and the emitter need no state).
+#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
+pub struct DInstr {
+ pub op: DOp,
+ pub dst: u8,
+ pub src: u8,
+ /// The third register of `AndX` and `OrX`; 0 on every other form (drawn and unused).
+ pub src2: u8,
+ /// The rotation 1..31 of `Rot` and `XRot`, else 0.
+ pub rot: u8,
+ /// The constant of `AddC`, `XorC` (any), `MulC` and `MulC2` (odd); 0 on every other form.
+ pub imm: u32,
+}
+
+/// The nine round programs of an item for one day, with the attempt that passed the acceptance test.
+#[derive(Clone, Debug, PartialEq, Eq)]
+pub struct DeriveProgram {
+ pub len: u32,
+ pub attempt: u32,
+ pub rounds: Vec>,
+}
+
+/// Why a candidate was rejected.
+#[derive(Clone, Debug, PartialEq, Eq)]
+pub enum DeriveReject {
+ /// A register no instruction of round program `round` writes.
+ RegisterNeverWritten { round: usize, reg: u8 },
+ /// Fewer than [`DISTINCT_ROTS_FLOOR`] distinct rotation amounts over the item's programs.
+ RotationsDegenerate { distinct: usize },
+ /// Chip operations per item under [`OPS_FLOOR_X8`] scaled to the length.
+ OpsUnderFloor { ops: u64, floor: u64 },
+ /// GPU operations per item under [`GPU_OPS_FLOOR_X8`] scaled to the length.
+ GpuOpsUnderFloor { ops: u64, floor: u64 },
+ /// Multiplies per item under [`MULS_FLOOR_X8`] scaled to the length.
+ MulsUnderFloor { muls: u64, floor: u64 },
+}
+
+impl std::fmt::Display for DeriveReject {
+ fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+ match self {
+ DeriveReject::RegisterNeverWritten { round, reg } => write!(f, "register {reg} never written in round program {round}"),
+ DeriveReject::RotationsDegenerate { distinct } => write!(f, "only {distinct} distinct rotation amounts"),
+ DeriveReject::OpsUnderFloor { ops, floor } => write!(f, "{ops} chip operations per item, floor {floor}"),
+ DeriveReject::GpuOpsUnderFloor { ops, floor } => write!(f, "{ops} GPU operations per item, floor {floor}"),
+ DeriveReject::MulsUnderFloor { muls, floor } => write!(f, "{muls} multiplies per item, floor {floor}"),
+ }
+ }
+}
+
+impl DeriveProgram {
+ /// Draw one candidate of `len` instructions per round program from `rng` (four draws per instruction).
+ pub fn draw_candidate(rng: &mut SplitMix64, len: u32, attempt: u32) -> DeriveProgram {
+ let mut rounds = Vec::with_capacity(DERIVE_PROGRAMS);
+ for _ in 0..DERIVE_PROGRAMS {
+ let mut prog = Vec::with_capacity(len as usize);
+ let mut chain = 0u8;
+ for _ in 0..len {
+ let op = DOp::for_roll(rng.below(100));
+ // the destination: the 15 registers other than the chain, in ascending order
+ let d_roll = rng.below((DERIVE_REGS - 1) as u64) as u8;
+ let dst = if d_roll >= chain { d_roll + 1 } else { d_roll };
+ // the third register: the 14 registers other than dst and the chain, in ascending order
+ let b_roll = rng.below((DERIVE_REGS - 2) as u64) as u8;
+ let (lo, hi) = if dst < chain { (dst, chain) } else { (chain, dst) };
+ let mut b = b_roll;
+ if b >= lo {
+ b += 1;
+ }
+ if b >= hi {
+ b += 1;
+ }
+ let x = rng.next() as u32;
+ let mut ins = DInstr { op, dst, src: chain, src2: 0, rot: 0, imm: 0 };
+ if op.has_third() {
+ ins.src2 = b;
+ }
+ if op.has_rot() {
+ ins.rot = 1 + (x % 31) as u8;
+ }
+ if op.has_imm() {
+ ins.imm = if matches!(op, DOp::MulC | DOp::MulC2) { x | 1 } else { x };
+ }
+ prog.push(ins);
+ chain = dst;
+ }
+ rounds.push(prog);
+ }
+ DeriveProgram { len, attempt, rounds }
+ }
+
+ /// Draw the program of a day: candidates from `rng` in turn until one passes [`DeriveProgram::check`].
+ /// Panics after [`MAX_ATTEMPTS`] (the floors sit more than 7 standard deviations under the expected counts, so
+ /// a rejection is a rare event and 64 in a row is not one that happens).
+ pub fn draw(rng: &mut SplitMix64, len: u32) -> DeriveProgram {
+ for attempt in 0..MAX_ATTEMPTS {
+ let p = Self::draw_candidate(rng, len, attempt);
+ if p.check().is_ok() {
+ return p;
+ }
+ }
+ panic!("derivation program: {MAX_ATTEMPTS} candidates rejected in a row");
+ }
+
+ /// The floors for this length (chip operations, GPU operations, multiplies): the x8 floors scaled by
+ /// `len / DERIVE_LEN_X8`, so a shorter class, measured as a fallback, has its own proportional floors.
+ pub fn floors(len: u32) -> (u64, u64, u64) {
+ let scale = |f: u64| f * len as u64 / DERIVE_LEN_X8 as u64;
+ (scale(OPS_FLOOR_X8), scale(GPU_OPS_FLOOR_X8), scale(MULS_FLOOR_X8))
+ }
+
+ /// The acceptance test: every register written in every round program; at least [`DISTINCT_ROTS_FLOOR`]
+ /// distinct rotation amounts; chip operations, GPU operations and multiplies per item at or above the floors
+ /// (the x8 mixer's counts from the code). The structural
+ /// rules (`dst != src`, the third register distinct, rotations in 1..31, odd multiplier constants) hold by
+ /// construction and are asserted.
+ pub fn check(&self) -> Result<(), DeriveReject> {
+ let mut rots = [false; 32];
+ for (r, prog) in self.rounds.iter().enumerate() {
+ let mut written = [false; DERIVE_REGS];
+ let mut chain = 0u8;
+ for ins in prog {
+ assert!(ins.src == chain && ins.dst != ins.src && (ins.dst as usize) < DERIVE_REGS, "chain rule");
+ if ins.op.has_third() {
+ assert!(ins.src2 != ins.dst && ins.src2 != ins.src && (ins.src2 as usize) < DERIVE_REGS, "third register");
+ }
+ if ins.op.has_rot() {
+ assert!((1..=31).contains(&ins.rot), "rotation");
+ rots[ins.rot as usize] = true;
+ }
+ if matches!(ins.op, DOp::MulC | DOp::MulC2) {
+ assert!(ins.imm & 1 == 1, "odd multiplier");
+ }
+ written[ins.dst as usize] = true;
+ chain = ins.dst;
+ }
+ if let Some(reg) = written.iter().position(|w| !w) {
+ return Err(DeriveReject::RegisterNeverWritten { round: r, reg: reg as u8 });
+ }
+ }
+ let distinct = rots.iter().filter(|r| **r).count();
+ if distinct < DISTINCT_ROTS_FLOOR {
+ return Err(DeriveReject::RotationsDegenerate { distinct });
+ }
+ let (ops_floor, gpu_floor, muls_floor) = Self::floors(self.len);
+ let ops = self.chip_ops();
+ if ops < ops_floor {
+ return Err(DeriveReject::OpsUnderFloor { ops, floor: ops_floor });
+ }
+ let gpu = self.gpu_ops();
+ if gpu < gpu_floor {
+ return Err(DeriveReject::GpuOpsUnderFloor { ops: gpu, floor: gpu_floor });
+ }
+ let muls = self.muls();
+ if muls < muls_floor {
+ return Err(DeriveReject::MulsUnderFloor { muls, floor: muls_floor });
+ }
+ Ok(())
+ }
+
+ pub fn instr_count(&self) -> u64 {
+ self.rounds.iter().map(|p| p.len() as u64).sum()
+ }
+ pub fn gpu_ops(&self) -> u64 {
+ self.rounds.iter().flatten().map(|i| i.op.gpu_ops()).sum()
+ }
+ pub fn chip_ops(&self) -> u64 {
+ self.rounds.iter().flatten().map(|i| i.op.chip_ops()).sum()
+ }
+ pub fn muls(&self) -> u64 {
+ self.rounds.iter().flatten().filter(|i| i.op.is_mul()).count() as u64
+ }
+ /// Count per op, in [`DOP_WEIGHTS`] order.
+ pub fn op_counts(&self) -> [u64; 12] {
+ let mut c = [0u64; 12];
+ for i in self.rounds.iter().flatten() {
+ c[i.op as usize] += 1;
+ }
+ c
+ }
+ /// "add=887 sub=..." in weight order.
+ pub fn op_mix(&self) -> String {
+ let c = self.op_counts();
+ DOP_WEIGHTS.iter().map(|(o, _)| format!("{}={}", o.name(), c[*o as usize])).collect::>().join(" ")
+ }
+ /// FNV-1a 64 over the instruction stream (op, dst, src, src2, rot, imm as bytes): the program's fingerprint
+ /// for packs and logs.
+ pub fn fingerprint(&self) -> u64 {
+ let mut b = Vec::with_capacity(self.instr_count() as usize * 9);
+ for i in self.rounds.iter().flatten() {
+ b.push(i.op as u8);
+ b.push(i.dst);
+ b.push(i.src);
+ b.push(i.src2);
+ b.push(i.rot);
+ b.extend_from_slice(&i.imm.to_le_bytes());
+ }
+ crate::seed::fnv1a64(&b)
+ }
+}
+
+/// Lanes of the SoA interpreter: the verifier derives up to 32 distinct items per load (one per lane of the
+/// unit), so each instruction runs across 32 item states at once and the dispatch is paid once per 32 items.
+pub const SOA_LANES: usize = 32;
+
+/// The item states of a batch, word-major: `st[reg][lane]`.
+pub type SoaState = [[u32; SOA_LANES]; DERIVE_REGS];
+
+#[inline(always)]
+fn rotl(x: u32, n: u32) -> u32 {
+ x.rotate_left(n)
+}
+
+/// The twelve forms over a batch, one function each, every one a straight loop over the lanes the compiler
+/// vectorises. The destination row and the source rows are distinct by the chain rule (`dst != src`, and the third
+/// register distinct from both: asserted by [`DeriveProgram::check`] and checked here in debug builds), so the
+/// rows are addressed through raw pointers rather than copied out of the state.
+mod forms {
+ use super::{rotl, DInstr, SoaState, SOA_LANES};
+ #[inline(always)]
+ pub fn add(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src);
+
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+
+ *d = d.wrapping_add(*c);
+ }
+ }
+ }
+ #[inline(always)]
+ pub fn sub(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src);
+
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+
+ *d = d.wrapping_sub(*c);
+ }
+ }
+ }
+ #[inline(always)]
+ pub fn xor(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src);
+
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+
+ *d ^= *c;
+ }
+ }
+ }
+ #[inline(always)]
+ pub fn mul(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src);
+
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+
+ *d = d.wrapping_mul(*c | 1);
+ }
+ }
+ }
+ #[inline(always)]
+ pub fn rot(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src);
+ let r = ins.rot as u32;
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+
+ *d = rotl(*d, r).wrapping_add(*c);
+ }
+ }
+ }
+ #[inline(always)]
+ pub fn xrot(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src);
+ let r = ins.rot as u32;
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+
+ *d = rotl(*d ^ *c, r);
+ }
+ }
+ }
+ #[inline(always)]
+ pub fn addc(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src);
+ let i = ins.imm;
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+
+ *d = d.wrapping_add(c.wrapping_add(i));
+ }
+ }
+ }
+ #[inline(always)]
+ pub fn xorc(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src);
+ let i = ins.imm;
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+
+ *d ^= *c ^ i;
+ }
+ }
+ }
+ #[inline(always)]
+ pub fn mulc(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src);
+ let i = ins.imm;
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+
+ *d = (*d ^ *c).wrapping_mul(i);
+ }
+ }
+ }
+ #[inline(always)]
+ pub fn mulc2(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src);
+ let i = ins.imm;
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+
+ *d = d.wrapping_mul(i).wrapping_add(*c);
+ }
+ }
+ }
+ #[inline(always)]
+ pub fn andx(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src && ins.src2 != ins.dst && ins.src2 != ins.src);
+
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+ let bp = st.as_ptr().add(ins.src2 as usize) as *const u32;
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+ let b = &*bp.add(k);
+ *d ^= *c & *b;
+ }
+ }
+ }
+ #[inline(always)]
+ pub fn orx(ins: &DInstr, st: &mut SoaState) {
+ debug_assert!(ins.dst != ins.src && ins.src2 != ins.dst && ins.src2 != ins.src);
+
+ // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
+ // pointers stay inside `st`.
+ unsafe {
+ let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
+ let cp = st.as_ptr().add(ins.src as usize) as *const u32;
+ let bp = st.as_ptr().add(ins.src2 as usize) as *const u32;
+ for k in 0..SOA_LANES {
+ let d = &mut *dp.add(k);
+ let c = &*cp.add(k);
+ let b = &*bp.add(k);
+ *d = d.wrapping_add(*c | *b);
+ }
+ }
+ }
+}
+
+/// One instruction over the batch (the single dispatch; [`run_round`] dispatches on pairs).
+#[inline(always)]
+pub fn run_instr(ins: &DInstr, st: &mut SoaState) {
+ match ins.op {
+ DOp::Add => forms::add(ins, st),
+ DOp::Sub => forms::sub(ins, st),
+ DOp::Xor => forms::xor(ins, st),
+ DOp::Mul => forms::mul(ins, st),
+ DOp::Rot => forms::rot(ins, st),
+ DOp::XRot => forms::xrot(ins, st),
+ DOp::AddC => forms::addc(ins, st),
+ DOp::XorC => forms::xorc(ins, st),
+ DOp::MulC => forms::mulc(ins, st),
+ DOp::MulC2 => forms::mulc2(ins, st),
+ DOp::AndX => forms::andx(ins, st),
+ DOp::OrX => forms::orx(ins, st),
+ }
+}
+
+/// Run one round program over the batch. The dispatch is on PAIRS of instructions (144 arms, one indirect branch
+/// per two instructions): the op sequence of a drawn program is random, so the branch predictor misses most
+/// dispatches, and the miss (about 3.7 ns of the 7.2 ns an instruction cost per batch on one M5 Max core, measured
+/// with `examples/derive_perf.rs` on 6 October 2026, a functional run) is paid once per pair instead of once per
+/// instruction. The result is bit for bit that of [`run_instr`] in sequence.
+#[inline(never)]
+pub fn run_round(prog: &[DInstr], st: &mut SoaState) {
+ let mut it = prog.chunks_exact(2);
+ for pair in &mut it {
+ let (a, b) = (&pair[0], &pair[1]);
+ match (a.op as u8) * 12 + b.op as u8 {
+ 0 => { forms::add(a, st); forms::add(b, st); }
+ 1 => { forms::add(a, st); forms::sub(b, st); }
+ 2 => { forms::add(a, st); forms::xor(b, st); }
+ 3 => { forms::add(a, st); forms::mul(b, st); }
+ 4 => { forms::add(a, st); forms::rot(b, st); }
+ 5 => { forms::add(a, st); forms::xrot(b, st); }
+ 6 => { forms::add(a, st); forms::addc(b, st); }
+ 7 => { forms::add(a, st); forms::xorc(b, st); }
+ 8 => { forms::add(a, st); forms::mulc(b, st); }
+ 9 => { forms::add(a, st); forms::mulc2(b, st); }
+ 10 => { forms::add(a, st); forms::andx(b, st); }
+ 11 => { forms::add(a, st); forms::orx(b, st); }
+ 12 => { forms::sub(a, st); forms::add(b, st); }
+ 13 => { forms::sub(a, st); forms::sub(b, st); }
+ 14 => { forms::sub(a, st); forms::xor(b, st); }
+ 15 => { forms::sub(a, st); forms::mul(b, st); }
+ 16 => { forms::sub(a, st); forms::rot(b, st); }
+ 17 => { forms::sub(a, st); forms::xrot(b, st); }
+ 18 => { forms::sub(a, st); forms::addc(b, st); }
+ 19 => { forms::sub(a, st); forms::xorc(b, st); }
+ 20 => { forms::sub(a, st); forms::mulc(b, st); }
+ 21 => { forms::sub(a, st); forms::mulc2(b, st); }
+ 22 => { forms::sub(a, st); forms::andx(b, st); }
+ 23 => { forms::sub(a, st); forms::orx(b, st); }
+ 24 => { forms::xor(a, st); forms::add(b, st); }
+ 25 => { forms::xor(a, st); forms::sub(b, st); }
+ 26 => { forms::xor(a, st); forms::xor(b, st); }
+ 27 => { forms::xor(a, st); forms::mul(b, st); }
+ 28 => { forms::xor(a, st); forms::rot(b, st); }
+ 29 => { forms::xor(a, st); forms::xrot(b, st); }
+ 30 => { forms::xor(a, st); forms::addc(b, st); }
+ 31 => { forms::xor(a, st); forms::xorc(b, st); }
+ 32 => { forms::xor(a, st); forms::mulc(b, st); }
+ 33 => { forms::xor(a, st); forms::mulc2(b, st); }
+ 34 => { forms::xor(a, st); forms::andx(b, st); }
+ 35 => { forms::xor(a, st); forms::orx(b, st); }
+ 36 => { forms::mul(a, st); forms::add(b, st); }
+ 37 => { forms::mul(a, st); forms::sub(b, st); }
+ 38 => { forms::mul(a, st); forms::xor(b, st); }
+ 39 => { forms::mul(a, st); forms::mul(b, st); }
+ 40 => { forms::mul(a, st); forms::rot(b, st); }
+ 41 => { forms::mul(a, st); forms::xrot(b, st); }
+ 42 => { forms::mul(a, st); forms::addc(b, st); }
+ 43 => { forms::mul(a, st); forms::xorc(b, st); }
+ 44 => { forms::mul(a, st); forms::mulc(b, st); }
+ 45 => { forms::mul(a, st); forms::mulc2(b, st); }
+ 46 => { forms::mul(a, st); forms::andx(b, st); }
+ 47 => { forms::mul(a, st); forms::orx(b, st); }
+ 48 => { forms::rot(a, st); forms::add(b, st); }
+ 49 => { forms::rot(a, st); forms::sub(b, st); }
+ 50 => { forms::rot(a, st); forms::xor(b, st); }
+ 51 => { forms::rot(a, st); forms::mul(b, st); }
+ 52 => { forms::rot(a, st); forms::rot(b, st); }
+ 53 => { forms::rot(a, st); forms::xrot(b, st); }
+ 54 => { forms::rot(a, st); forms::addc(b, st); }
+ 55 => { forms::rot(a, st); forms::xorc(b, st); }
+ 56 => { forms::rot(a, st); forms::mulc(b, st); }
+ 57 => { forms::rot(a, st); forms::mulc2(b, st); }
+ 58 => { forms::rot(a, st); forms::andx(b, st); }
+ 59 => { forms::rot(a, st); forms::orx(b, st); }
+ 60 => { forms::xrot(a, st); forms::add(b, st); }
+ 61 => { forms::xrot(a, st); forms::sub(b, st); }
+ 62 => { forms::xrot(a, st); forms::xor(b, st); }
+ 63 => { forms::xrot(a, st); forms::mul(b, st); }
+ 64 => { forms::xrot(a, st); forms::rot(b, st); }
+ 65 => { forms::xrot(a, st); forms::xrot(b, st); }
+ 66 => { forms::xrot(a, st); forms::addc(b, st); }
+ 67 => { forms::xrot(a, st); forms::xorc(b, st); }
+ 68 => { forms::xrot(a, st); forms::mulc(b, st); }
+ 69 => { forms::xrot(a, st); forms::mulc2(b, st); }
+ 70 => { forms::xrot(a, st); forms::andx(b, st); }
+ 71 => { forms::xrot(a, st); forms::orx(b, st); }
+ 72 => { forms::addc(a, st); forms::add(b, st); }
+ 73 => { forms::addc(a, st); forms::sub(b, st); }
+ 74 => { forms::addc(a, st); forms::xor(b, st); }
+ 75 => { forms::addc(a, st); forms::mul(b, st); }
+ 76 => { forms::addc(a, st); forms::rot(b, st); }
+ 77 => { forms::addc(a, st); forms::xrot(b, st); }
+ 78 => { forms::addc(a, st); forms::addc(b, st); }
+ 79 => { forms::addc(a, st); forms::xorc(b, st); }
+ 80 => { forms::addc(a, st); forms::mulc(b, st); }
+ 81 => { forms::addc(a, st); forms::mulc2(b, st); }
+ 82 => { forms::addc(a, st); forms::andx(b, st); }
+ 83 => { forms::addc(a, st); forms::orx(b, st); }
+ 84 => { forms::xorc(a, st); forms::add(b, st); }
+ 85 => { forms::xorc(a, st); forms::sub(b, st); }
+ 86 => { forms::xorc(a, st); forms::xor(b, st); }
+ 87 => { forms::xorc(a, st); forms::mul(b, st); }
+ 88 => { forms::xorc(a, st); forms::rot(b, st); }
+ 89 => { forms::xorc(a, st); forms::xrot(b, st); }
+ 90 => { forms::xorc(a, st); forms::addc(b, st); }
+ 91 => { forms::xorc(a, st); forms::xorc(b, st); }
+ 92 => { forms::xorc(a, st); forms::mulc(b, st); }
+ 93 => { forms::xorc(a, st); forms::mulc2(b, st); }
+ 94 => { forms::xorc(a, st); forms::andx(b, st); }
+ 95 => { forms::xorc(a, st); forms::orx(b, st); }
+ 96 => { forms::mulc(a, st); forms::add(b, st); }
+ 97 => { forms::mulc(a, st); forms::sub(b, st); }
+ 98 => { forms::mulc(a, st); forms::xor(b, st); }
+ 99 => { forms::mulc(a, st); forms::mul(b, st); }
+ 100 => { forms::mulc(a, st); forms::rot(b, st); }
+ 101 => { forms::mulc(a, st); forms::xrot(b, st); }
+ 102 => { forms::mulc(a, st); forms::addc(b, st); }
+ 103 => { forms::mulc(a, st); forms::xorc(b, st); }
+ 104 => { forms::mulc(a, st); forms::mulc(b, st); }
+ 105 => { forms::mulc(a, st); forms::mulc2(b, st); }
+ 106 => { forms::mulc(a, st); forms::andx(b, st); }
+ 107 => { forms::mulc(a, st); forms::orx(b, st); }
+ 108 => { forms::mulc2(a, st); forms::add(b, st); }
+ 109 => { forms::mulc2(a, st); forms::sub(b, st); }
+ 110 => { forms::mulc2(a, st); forms::xor(b, st); }
+ 111 => { forms::mulc2(a, st); forms::mul(b, st); }
+ 112 => { forms::mulc2(a, st); forms::rot(b, st); }
+ 113 => { forms::mulc2(a, st); forms::xrot(b, st); }
+ 114 => { forms::mulc2(a, st); forms::addc(b, st); }
+ 115 => { forms::mulc2(a, st); forms::xorc(b, st); }
+ 116 => { forms::mulc2(a, st); forms::mulc(b, st); }
+ 117 => { forms::mulc2(a, st); forms::mulc2(b, st); }
+ 118 => { forms::mulc2(a, st); forms::andx(b, st); }
+ 119 => { forms::mulc2(a, st); forms::orx(b, st); }
+ 120 => { forms::andx(a, st); forms::add(b, st); }
+ 121 => { forms::andx(a, st); forms::sub(b, st); }
+ 122 => { forms::andx(a, st); forms::xor(b, st); }
+ 123 => { forms::andx(a, st); forms::mul(b, st); }
+ 124 => { forms::andx(a, st); forms::rot(b, st); }
+ 125 => { forms::andx(a, st); forms::xrot(b, st); }
+ 126 => { forms::andx(a, st); forms::addc(b, st); }
+ 127 => { forms::andx(a, st); forms::xorc(b, st); }
+ 128 => { forms::andx(a, st); forms::mulc(b, st); }
+ 129 => { forms::andx(a, st); forms::mulc2(b, st); }
+ 130 => { forms::andx(a, st); forms::andx(b, st); }
+ 131 => { forms::andx(a, st); forms::orx(b, st); }
+ 132 => { forms::orx(a, st); forms::add(b, st); }
+ 133 => { forms::orx(a, st); forms::sub(b, st); }
+ 134 => { forms::orx(a, st); forms::xor(b, st); }
+ 135 => { forms::orx(a, st); forms::mul(b, st); }
+ 136 => { forms::orx(a, st); forms::rot(b, st); }
+ 137 => { forms::orx(a, st); forms::xrot(b, st); }
+ 138 => { forms::orx(a, st); forms::addc(b, st); }
+ 139 => { forms::orx(a, st); forms::xorc(b, st); }
+ 140 => { forms::orx(a, st); forms::mulc(b, st); }
+ 141 => { forms::orx(a, st); forms::mulc2(b, st); }
+ 142 => { forms::orx(a, st); forms::andx(b, st); }
+ 143 => { forms::orx(a, st); forms::orx(b, st); }
+ _ => unreachable!(),
+ }
+ }
+ for ins in it.remainder() {
+ run_instr(ins, st);
+ }
+}
+
+/// The scalar reference: one instruction on one 16-word state, the text the kernels carry (`emit.rs`,
+/// `derive_instr_text`) restated in Rust. The tests pin the SoA interpreter against it.
+pub fn run_round_scalar(prog: &[DInstr], s: &mut [u32; DERIVE_REGS]) {
+ for ins in prog {
+ let d = ins.dst as usize;
+ let c = s[ins.src as usize];
+ match ins.op {
+ DOp::Add => s[d] = s[d].wrapping_add(c),
+ DOp::Sub => s[d] = s[d].wrapping_sub(c),
+ DOp::Xor => s[d] ^= c,
+ DOp::Mul => s[d] = s[d].wrapping_mul(c | 1),
+ DOp::Rot => s[d] = rotl(s[d], ins.rot as u32).wrapping_add(c),
+ DOp::XRot => s[d] = rotl(s[d] ^ c, ins.rot as u32),
+ DOp::AddC => s[d] = s[d].wrapping_add(c.wrapping_add(ins.imm)),
+ DOp::XorC => s[d] ^= c ^ ins.imm,
+ DOp::MulC => s[d] = (s[d] ^ c).wrapping_mul(ins.imm),
+ DOp::MulC2 => s[d] = s[d].wrapping_mul(ins.imm).wrapping_add(c),
+ DOp::AndX => s[d] ^= c & s[ins.src2 as usize],
+ DOp::OrX => s[d] = s[d].wrapping_add(c | s[ins.src2 as usize]),
+ }
+ }
+}
+
+/// The source text of one instruction in the C-family dialects (the same text in Metal, CUDA C and OpenCL C:
+/// `s` is the 16-word state, `mh_rotl` the rotate of the memhard core).
+pub fn instr_text(ins: &DInstr) -> String {
+ let (d, c, b) = (ins.dst, ins.src, ins.src2);
+ match ins.op {
+ DOp::Add => format!("s[{d}] += s[{c}];"),
+ DOp::Sub => format!("s[{d}] -= s[{c}];"),
+ DOp::Xor => format!("s[{d}] ^= s[{c}];"),
+ DOp::Mul => format!("s[{d}] *= (s[{c}] | 1u);"),
+ DOp::Rot => format!("s[{d}] = mh_rotl(s[{d}], {}u) + s[{c}];", ins.rot),
+ DOp::XRot => format!("s[{d}] = mh_rotl(s[{d}] ^ s[{c}], {}u);", ins.rot),
+ DOp::AddC => format!("s[{d}] += s[{c}] + {:#010x}u;", ins.imm),
+ DOp::XorC => format!("s[{d}] ^= s[{c}] ^ {:#010x}u;", ins.imm),
+ DOp::MulC => format!("s[{d}] = (s[{d}] ^ s[{c}]) * {:#010x}u;", ins.imm),
+ DOp::MulC2 => format!("s[{d}] = s[{d}] * {:#010x}u + s[{c}];", ins.imm),
+ DOp::AndX => format!("s[{d}] ^= (s[{c}] & s[{b}]);"),
+ DOp::OrX => format!("s[{d}] += (s[{c}] | s[{b}]);"),
+ }
+}
+
+/// One instruction as a line of program.json: `"add d=3 c=0"`, `"mulc d=5 c=3 imm=0x..."`.
+pub fn instr_line(ins: &DInstr) -> String {
+ let mut s = format!("{} d={} c={}", ins.op.name(), ins.dst, ins.src);
+ if ins.op.has_third() {
+ s.push_str(&format!(" b={}", ins.src2));
+ }
+ if ins.op.has_rot() {
+ s.push_str(&format!(" k={}", ins.rot));
+ }
+ if ins.op.has_imm() {
+ s.push_str(&format!(" imm={:#010x}", ins.imm));
+ }
+ s
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+
+ #[test]
+ fn weights_sum_and_ops() {
+ assert_eq!(DOP_WEIGHTS.iter().map(|(_, w)| w).sum::(), 100);
+ for (i, (op, _)) in DOP_WEIGHTS.iter().enumerate() {
+ assert_eq!(*op as usize, i);
+ assert_eq!(DOp::from_u8(i as u8), Some(*op));
+ }
+ assert_eq!(DOp::for_roll(0), DOp::Add);
+ assert_eq!(DOp::for_roll(13), DOp::Add);
+ assert_eq!(DOp::for_roll(14), DOp::Sub);
+ assert_eq!(DOp::for_roll(99), DOp::OrX);
+ // the expected chip operations per instruction, 1.52 (1.62 on a GPU), put 736 x 9 over the x8 floors
+ let mean: f64 = DOP_WEIGHTS.iter().map(|(o, w)| o.chip_ops() as f64 * *w as f64 / 100.0).sum();
+ assert!((mean - 1.52).abs() < 1e-9, "{mean}");
+ let gpu_mean: f64 = DOP_WEIGHTS.iter().map(|(o, w)| o.gpu_ops() as f64 * *w as f64 / 100.0).sum();
+ assert!((gpu_mean - 1.62).abs() < 1e-9, "{gpu_mean}");
+ assert!(9.0 * DERIVE_LEN_X8 as f64 * mean > OPS_FLOOR_X8 as f64);
+ assert!(9.0 * DERIVE_LEN_X8 as f64 * gpu_mean > GPU_OPS_FLOOR_X8 as f64);
+ assert_eq!(DeriveProgram::floors(DERIVE_LEN_X8), (9_216, 10_368, 1_152));
+ assert_eq!(DeriveProgram::floors(368), (4_608, 5_184, 576));
+ let mul_share: f64 = DOP_WEIGHTS.iter().filter(|(o, _)| o.is_mul()).map(|(_, w)| *w as f64 / 100.0).sum();
+ assert!(9.0 * DERIVE_LEN_X8 as f64 * mul_share > MULS_FLOOR_X8 as f64);
+ }
+
+ #[test]
+ fn draw_is_structural_and_accepted() {
+ let mut rng = SplitMix64::new(0x1234_5678_9abc_def0);
+ let p = DeriveProgram::draw(&mut rng, DERIVE_LEN_X8);
+ assert_eq!(p.attempt, 0, "the first candidate of this seed passes");
+ assert_eq!(p.rounds.len(), DERIVE_PROGRAMS);
+ assert_eq!(p.instr_count(), 9 * DERIVE_LEN_X8 as u64);
+ assert!(p.check().is_ok());
+ assert!(p.chip_ops() >= OPS_FLOOR_X8 && p.gpu_ops() >= GPU_OPS_FLOOR_X8 && p.muls() >= MULS_FLOOR_X8);
+ assert!(p.gpu_ops() > p.chip_ops());
+ // every instruction consumes the newest result
+ for prog in &p.rounds {
+ let mut chain = 0u8;
+ for ins in prog {
+ assert_eq!(ins.src, chain);
+ assert_ne!(ins.dst, chain);
+ chain = ins.dst;
+ }
+ }
+ // four draws per instruction: the same program again from the same seed, and a different one one draw on
+ let mut rng2 = SplitMix64::new(0x1234_5678_9abc_def0);
+ assert_eq!(DeriveProgram::draw(&mut rng2, DERIVE_LEN_X8), p);
+ let mut rng3 = SplitMix64::new(0x1234_5678_9abc_def0);
+ rng3.next();
+ assert_ne!(DeriveProgram::draw(&mut rng3, DERIVE_LEN_X8), p);
+ }
+
+ #[test]
+ fn soa_matches_scalar_and_is_a_bijection() {
+ let mut rng = SplitMix64::new(7);
+ let p = DeriveProgram::draw(&mut rng, 64);
+ let mut st: SoaState = [[0u32; SOA_LANES]; DERIVE_REGS];
+ let mut scalars = [[0u32; DERIVE_REGS]; SOA_LANES];
+ let mut x = SplitMix64::new(99);
+ for k in 0..SOA_LANES {
+ for r in 0..DERIVE_REGS {
+ let v = x.next() as u32;
+ st[r][k] = v;
+ scalars[k][r] = v;
+ }
+ }
+ let before = scalars;
+ for prog in &p.rounds {
+ run_round(prog, &mut st);
+ for k in 0..SOA_LANES {
+ run_round_scalar(prog, &mut scalars[k]);
+ }
+ }
+ for k in 0..SOA_LANES {
+ for r in 0..DERIVE_REGS {
+ assert_eq!(st[r][k], scalars[k][r], "lane {k} reg {r}");
+ }
+ }
+ // distinct inputs stay distinct (a bijection on the state, spot-checked: 32 lanes, no collision)
+ for a in 0..SOA_LANES {
+ for b in a + 1..SOA_LANES {
+ assert_ne!(scalars[a], scalars[b]);
+ assert_ne!(before[a], before[b]);
+ }
+ }
+ }
+
+ #[test]
+ fn acceptance_rejects_degenerate_draws() {
+ let mut rng = SplitMix64::new(3);
+ let mut p = DeriveProgram::draw(&mut rng, 64);
+ // a register never written: make every write of round 2 go to the chain's neighbour
+ let mut q = p.clone();
+ for ins in q.rounds[2].iter_mut() {
+ ins.dst = if ins.src == 1 { 2 } else { 1 };
+ }
+ let mut chain = 0u8;
+ for ins in q.rounds[2].iter_mut() {
+ ins.src = chain;
+ ins.dst = if chain == 1 { 2 } else { 1 };
+ chain = ins.dst;
+ }
+ assert!(matches!(q.check(), Err(DeriveReject::RegisterNeverWritten { round: 2, .. })));
+ // all rotations equal
+ let mut q = p.clone();
+ for ins in q.rounds.iter_mut().flatten() {
+ if ins.op.has_rot() {
+ ins.rot = 5;
+ }
+ }
+ assert!(matches!(q.check(), Err(DeriveReject::RotationsDegenerate { distinct: 1 })));
+ // every op an add apart from the xor-rotates (so the rotations stay distinct): under the ops floor
+ for ins in p.rounds.iter_mut().flatten() {
+ if ins.op != DOp::XRot {
+ ins.op = DOp::Add;
+ ins.rot = 0;
+ ins.imm = 0;
+ ins.src2 = 0;
+ }
+ }
+ assert!(matches!(p.check(), Err(DeriveReject::OpsUnderFloor { .. })), "{:?}", p.check());
+ // no multiplies at all but the ops floor met: under the multiply floor
+ let mut q = DeriveProgram::draw(&mut SplitMix64::new(11), 64);
+ for ins in q.rounds.iter_mut().flatten() {
+ if ins.op.is_mul() {
+ ins.op = DOp::AddC;
+ }
+ }
+ assert!(matches!(q.check(), Err(DeriveReject::MulsUnderFloor { .. })), "{:?}", q.check());
+ }
+
+ #[test]
+ fn text_forms() {
+ let i = DInstr { op: DOp::MulC, dst: 5, src: 3, src2: 0, rot: 0, imm: 0x9e37_79b9 };
+ assert_eq!(instr_text(&i), "s[5] = (s[5] ^ s[3]) * 0x9e3779b9u;");
+ assert_eq!(instr_line(&i), "mulc d=5 c=3 imm=0x9e3779b9");
+ let i = DInstr { op: DOp::XRot, dst: 0, src: 15, src2: 0, rot: 17, imm: 0 };
+ assert_eq!(instr_text(&i), "s[0] = mh_rotl(s[0] ^ s[15], 17u);");
+ let i = DInstr { op: DOp::AndX, dst: 2, src: 9, src2: 14, rot: 0, imm: 0 };
+ assert_eq!(instr_text(&i), "s[2] ^= (s[9] & s[14]);");
+ }
+}
diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/emit.rs b/tools/attack/adv-accept-v5/igneum-pow/src/emit.rs
new file mode 100644
index 000000000..90105ba01
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/src/emit.rs
@@ -0,0 +1,2316 @@
+//! Kernel source emitters. Since 4 October 2026 (generator version 2) this crate is the source of every pack in
+//! `proto-cuda/packs/`; the pack tests diff the emitters against the checked-in files. Each function started as a
+//! byte-for-byte twin of its namesake in `proto-metal/main.swift` (`generateMSL`, `memhardMSL`, `emitMemhardCore`,
+//! `generateCUDA`, `generateOpenCL`, `generateProgramHeader`, `generateMemhardHeader`, `generateVectorsHeader`,
+//! `generateProgramJSON`, `generateVectorsJSON`); the kernel text is unchanged by version 2, and `program.json` and
+//! `program.h` carry the generator version, the attempt and the program id so no version 1 pack can be mistaken
+//! for a current one.
+//!
+//! One deliberate difference from the Swift: `program_json` writes the cache line mask inside the `"item"` string
+//! as a bare `0x003fffff`. The Swift writes it quoted (`jhex`), which is not valid JSON.
+
+use crate::generator::{EraParams, Instr, Op, Program, ProgramClass, GENERATOR_VERSION, INSTR_COUNT, ITERATIONS, LOAD_SLOTS, PROGRAM_SUBVERSION_V4};
+use crate::derive::{instr_line as derive_instr_line, instr_text as derive_instr_text, DERIVE_PROGRAMS};
+use crate::memhard::{
+ hot_key, hot_segments, hot_words, Layout, MixParams, Shape, CACHE_LINES_PER_SEGMENT, CACHE_SEGMENT_LOG2_LINES, CACHE_TAG,
+ CHACHA_ROUNDS, CHACHA_SIGMA, HOT_TAG, ITEM_ROUNDS,
+};
+use crate::seed::SplitMix64;
+use crate::verify::{window, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2, FOLD_MUL, FOLD_ROT};
+
+/// The index expression of a dataset load (era layout, `docs/plans/era-layout.md` section 1.3). For every class
+/// without an era it is the lottery hash's `rN & MASK`; for an era program it is the one form
+/// `((rotl_imm(rN * M, R) & WM) | OFF) & MASK` with the site's window constants at the pack's dataset size.
+fn load_index_expr(dialect: CoreDialect, era: Option<&EraParams>, ins: &Instr, a: &str, dataset_log2: u32) -> String {
+ let mask_name = match dialect {
+ CoreDialect::Metal => "MASK",
+ _ => "mask",
+ };
+ match era {
+ None => format!("{a} & {mask_name}"),
+ Some(e) => {
+ let (wm, off) = window(ins, mask_for(dataset_log2), dataset_log2);
+ format!("((rotl_imm({a} * {}, {}u) & {}) | {}) & {mask_name}", hex(e.stride_mul), e.stride_rot, hex(wm), hex(off))
+ }
+ }
+}
+
+/// The era lines of program.h (empty without an era).
+fn era_header_lines(p: &Program) -> String {
+ let Some(e) = p.class.era else { return String::new() };
+ let mut s = String::new();
+ s.push_str("// Era layout (5 October 2026, docs/plans/era-layout.md): NOT the lottery hash. Every dataset load reads\n");
+ s.push_str("// idx = ((rotl(src * STRIDE_MUL, STRIDE_ROT) & window mask) | window offset) & MASK; the window of a load site is the\n");
+ s.push_str("// dataset, a half or a quarter of it (IGNEUM_ERA_WINDOWS: site:shrink:offset); dataset word w holds word j(w) of item\n");
+ s.push_str("// t(w) with j's bits at the INTERLEAVE positions (memhard.h: mh_t, mh_j, mh_addr).\n");
+ s.push_str(&format!("#define IGNEUM_ERA_LABEL {}\n", jstr(&e.label())));
+ s.push_str(&format!("#define IGNEUM_ERA_SEED_WORDS {{ {} }}\n", join_hex(&e.words)));
+ s.push_str(&format!("#define IGNEUM_ERA_ALLOWED_WIDTHS {{ {}, {}, {} }} // words, ascending, 0 = unused; one entry pins the width\n", e.allowed[0], e.allowed[1], e.allowed[2]));
+ s.push_str(&format!("#define IGNEUM_ERA_WIDTH_WORDS {}\n", e.width_words));
+ s.push_str(&format!("#define IGNEUM_ERA_STRIDE_MUL {}\n", hex(e.stride_mul)));
+ s.push_str(&format!("#define IGNEUM_ERA_STRIDE_ROT {}\n", e.stride_rot));
+ s.push_str(&format!("#define IGNEUM_ERA_INTERLEAVE {{ {}, {}, {}, {} }}\n", e.pos[0], e.pos[1], e.pos[2], e.pos[3]));
+ s.push_str(&format!("#define IGNEUM_ERA_WINDOWS {}\n", jstr(&era_windows(p))));
+ s
+}
+
+/// "site:shrink:offset" for every load site of an era program, space separated.
+fn era_windows(p: &Program) -> String {
+ p.instrs
+ .iter()
+ .enumerate()
+ .filter(|(_, i)| i.op == Op::Load)
+ .map(|(k, i)| format!("{k}:{}:{}", i.win, i.off))
+ .collect::>()
+ .join(" ")
+}
+
+/// The layout helpers of the memory-hard core for a non-linear layout: `mh_j(w)`, `mh_t(w)` and `mh_addr(t, j)`
+/// (`Layout::split` and `Layout::join` as text). Empty for the linear layout, so the pinned packs do not change.
+fn layout_helpers(layout: Layout, u: &str, fn_: &str) -> String {
+ if layout.is_linear() {
+ return String::new();
+ }
+ let p = layout.pos;
+ let low = |q: u8| hex(((1u64 << q) - 1) as u32);
+ let mut s = String::new();
+ s.push_str(&format!(
+ "// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions {} {} {} {} of w.\n",
+ p[0], p[1], p[2], p[3]
+ ));
+ s.push_str(&format!(
+ "{fn_} {u} mh_j({u} w) {{ return ((w >> {}u) & 1u) | (((w >> {}u) & 1u) << 1) | (((w >> {}u) & 1u) << 2) | (((w >> {}u) & 1u) << 3); }}\n",
+ p[0], p[1], p[2], p[3]
+ ));
+ s.push_str(&format!("{fn_} {u} mh_t({u} w) {{"));
+ for &q in p.iter().rev() {
+ s.push_str(&format!(" w = (w & {}) | ((w >> {}u) << {}u);", low(q), q + 1, q));
+ }
+ s.push_str(" return w; }\n");
+ s.push_str(&format!("{fn_} {u} mh_addr({u} t, {u} j) {{ {u} w = t;"));
+ for (i, &q) in p.iter().enumerate() {
+ s.push_str(&format!(" w = ((w >> {}u) << {}u) | (w & {}) | (((j >> {}u) & 1u) << {}u);", q, q + 1, low(q), i, q));
+ }
+ s.push_str(" return w; }\n");
+ s
+}
+
+/// Where the words of a wide load come from (read-width experiment).
+#[derive(Clone, Copy, PartialEq, Eq)]
+enum WideSource {
+ /// `dataset`/`ds`: vector loads from the stored dataset.
+ Stored,
+ /// the closed form per word (Metal inline shortcut kernel).
+ InlineClosed,
+ /// one `mh_item` derivation per load, words taken from it (Metal inline memory-hard kernel).
+ InlineMemhard,
+}
+
+/// One wide `load` as a single statement block (read-width experiment, 5 October 2026): `width` words from the
+/// address aligned down to `width` words, folded into `dst` as `verify::fold_words`. The emitted text is the
+/// same shape in the three dialects: the vector loads differ (`uint4` pointer on Metal and CUDA, `vload4` on
+/// OpenCL C 1.2). For `width == 1` the caller emits the lottery hash's one-word form instead.
+fn wide_load_stmt(dialect: CoreDialect, d: &str, idx: &str, width: u8, src: WideSource, closed: Option<(u32, u32)>) -> String {
+ debug_assert!(width == 4 || width == 16);
+ let (u, base_ptr) = match dialect {
+ CoreDialect::Metal => ("uint", "dataset"),
+ CoreDialect::Cuda => ("uint32_t", "ds"),
+ CoreDialect::OpenCl => ("uint", "ds"),
+ };
+ let vectors = width as usize / 4;
+ let mut s = String::with_capacity(400);
+ s.push_str(&format!("{{ {u} b_ = ({idx}) & ~{}u; ", width as u32 - 1));
+ match src {
+ WideSource::Stored => match dialect {
+ CoreDialect::Metal => s.push_str(&format!("device const uint4* l_ = (device const uint4*)({base_ptr} + b_); ")),
+ CoreDialect::Cuda => s.push_str(&format!("const uint4* l_ = (const uint4*)({base_ptr} + b_); ")),
+ CoreDialect::OpenCl => {}
+ },
+ WideSource::InlineClosed => {}
+ WideSource::InlineMemhard => s.push_str("uint s_[16]; mh_item(cache, mh_t(b_), s_); "),
+ }
+ let word = |j: usize| -> String {
+ match src {
+ WideSource::Stored => format!("v{}_.{}", j / 4, ["x", "y", "z", "w"][j % 4]),
+ WideSource::InlineClosed => {
+ let (d0, d1) = closed.expect("closed-form words need d0, d1");
+ format!("ds_elem(b_ + {j}u, {}, {})", hex(d0), hex(d1))
+ }
+ WideSource::InlineMemhard => format!("s_[mh_j(b_) + {j}u]"),
+ }
+ };
+ if src == WideSource::Stored {
+ for v in 0..vectors {
+ match dialect {
+ CoreDialect::OpenCl => s.push_str(&format!("uint4 v{v}_ = vload4({v}u, {base_ptr} + b_); ")),
+ _ => s.push_str(&format!("uint4 v{v}_ = l_[{v}]; ")),
+ }
+ }
+ }
+ s.push_str(&format!("{u} x_ = {d} ^ {}; ", word(0)));
+ for j in 1..width as usize {
+ s.push_str(&format!("x_ = (rotl_imm(x_, {FOLD_ROT}u) * {}) ^ {}; ", hex(FOLD_MUL), word(j)));
+ }
+ s.push_str(&format!("{d} = x_; }}"));
+ s
+}
+
+/// The program class lines of program.h (Counter ASIC 2.0): `IGNEUM_PROGRAM_CLASS` and, when the program was drawn
+/// on the chain, `IGNEUM_ERA_SEED_HEX`. Empty for every version 2 program, so the pinned packs do not change; a
+/// worker reads an absent line as class v2. The generator version is `IGNEUM_GENERATOR` as before (3 for class v3,
+/// 4 for class v4; the v3 comment lines are byte for byte the 0.3.11 ones, so the pinned v3 packs do not change).
+fn program_class_header_lines(p: &Program) -> String {
+ if p.program_class() == ProgramClass::V2 {
+ return String::new();
+ }
+ let mut s = String::new();
+ if p.program_class() == ProgramClass::V5 {
+ s.push_str("// Program class v5 (proof of stored state and of following, docs/design/class-v5-stored-state.md): generator version 5,\n");
+ s.push_str("// class v4 over a dataset whose every item is keyed by the window's execution state (IGNEUM_STATE_* below, leaves.bin);\n");
+ s.push_str("// a worker that runs another class refuses this pack, and a job line names the class it wants (class=v5 era=).\n");
+ } else if p.program_class() == ProgramClass::V4 {
+ s.push_str("// Program class v4 (Counter ASIC 3.0, docs/plans/counter-asic-3-node.md): generator version 4, class v3 plus the\n");
+ s.push_str("// latency-shadow block (IGNEUM_SHADOW_INSTRS x IGNEUM_SHADOW_REPS per iteration); a worker that runs another class\n");
+ s.push_str("// refuses this pack, and a job line names the class it wants (class=v4 era=).\n");
+ } else {
+ s.push_str("// Program class v3 (Counter ASIC 2.0, docs/plans/counter-asic-2-rollout.md): generator version 3; a worker that\n");
+ s.push_str("// runs another class refuses this pack, and a job line names the class it wants (class=v3 era=).\n");
+ }
+ s.push_str(&format!("#define IGNEUM_PROGRAM_CLASS {}\n", jstr(p.program_class().name())));
+ if p.program_class() == ProgramClass::V4 {
+ // the class v4 stream sub-version (AP-F8-1 amendment): a worker ignores it, packcheck requires it
+ s.push_str(&format!("#define IGNEUM_PROGRAM_SUBVERSION {}\n", PROGRAM_SUBVERSION_V4));
+ }
+ if let Some(era) = &p.era_bytes {
+ s.push_str(&format!("#define IGNEUM_ERA_SEED_HEX {}\n", jstr(&hex_bytes(era))));
+ }
+ s
+}
+
+/// The state lines of program.h (class v5): the window's reference block and state root, the leaf count, the FNV of
+/// `leaves.bin` and the file's name. Empty for every dataset without leaves, so no pinned pack changes.
+fn state_header_lines(ds: &DatasetSource) -> String {
+ let Some(l) = ds.leaves() else { return String::new() };
+ let mut s = String::new();
+ s.push_str("// Class v5 state (docs/design/class-v5-stored-state.md): the window's reference chain block and the state root after it;\n");
+ s.push_str("// leaves.bin holds IGNEUM_STATE_LEAVES leaves of 16 little-endian words, leaf(t) = leaves[t mod IGNEUM_STATE_LEAVES].\n");
+ s.push_str(&format!("#define IGNEUM_STATE_BLOCK_HEX {}\n", jstr(&hex_bytes(&l.block))));
+ s.push_str(&format!("#define IGNEUM_STATE_BLOCK_NUMBER {}\n", l.number));
+ s.push_str(&format!("#define IGNEUM_STATE_ROOT_HEX {}\n", jstr(&hex_bytes(&l.root))));
+ s.push_str(&format!("#define IGNEUM_STATE_LEAVES {}\n", l.n()));
+ s.push_str(&format!("#define IGNEUM_STATE_RECORDS {}\n", l.records_total));
+ s.push_str(&format!("#define IGNEUM_STATE_SAMPLED {}\n", l.sampled as u8));
+ s.push_str(&format!("#define IGNEUM_STATE_LEAVES_FNV64 {}\n", hex64(l.fnv1a64())));
+ s.push_str("#define IGNEUM_STATE_LEAVES_FILE \"leaves.bin\"\n");
+ s
+}
+
+/// The load class lines of program.h (empty for the lottery hash, so the pinned packs do not change).
+fn class_header_lines(p: &Program) -> String {
+ if p.class.is_v2() {
+ return String::new();
+ }
+ let mut s = String::new();
+ if p.class.v2_loads() {
+ s.push_str("// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item
+");
+ s.push_str("// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule.
+");
+ } else {
+ s.push_str("// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads
+");
+ s.push_str("// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x.
+");
+ }
+ s.push_str(&format!("#define IGNEUM_LOAD_CLASS {}
+", jstr(&p.class.name())));
+ if p.class.mixer_mult != 1 || p.class.growth {
+ s.push_str(&format!("#define IGNEUM_CLASS_MIXER_MULT {}
+", p.class.mixer_mult));
+ s.push_str(&format!("#define IGNEUM_CACHE_GROWTH {} // 1: cache words = 2^(26 + doublings(day)), doublings = floor(log2(1 + day / 1460))
+", p.class.growth as u8));
+ }
+ if p.class.derive_len != 0 {
+ s.push_str("// Counter ASIC 3.0 item 2 (6 October 2026, docs/plans/counter-asic-3-derivation.md, a prototype, NOT class v3): the item
+");
+ s.push_str("// derivation runs the day's drawn program (memhard.h: mh_round_0..8, IGNEUM_DERIVE_LEN instructions each) in place of the mixer.
+");
+ s.push_str(&format!("#define IGNEUM_CLASS_DERIVE_LEN {}
+", p.class.derive_len));
+ }
+ s.push_str(&format!("#define IGNEUM_LOAD_SLOTS {}
+", p.class.load_slots));
+ s.push_str(&format!("#define IGNEUM_LOAD_MIX {{ {}, {}, {} }}
+", p.class.mix[0], p.class.mix[1], p.class.mix[2]));
+ let c = p.width_counts();
+ s.push_str(&format!("#define IGNEUM_LOAD_WIDTH_COUNTS {{ {}, {}, {} }} // loads of 4, 16, 64 bytes per program
+", c[0], c[1], c[2]));
+ s.push_str(&format!("#define IGNEUM_BYTES_PER_HASH {}
+", p.bytes_per_hash()));
+ s.push_str(&format!("#define IGNEUM_FOLD_ROT {FOLD_ROT}
+"));
+ s.push_str(&format!("#define IGNEUM_FOLD_MUL {}
+", hex(FOLD_MUL)));
+ s
+}
+
+/// The width of a load instruction's statement, for the emitters (1 for every non-load op).
+fn load_width(ins: &Instr) -> u8 {
+ if ins.op == Op::Load {
+ ins.width
+ } else {
+ 1
+ }
+}
+
+/// Variant 5 prelude: `scr_fill(gbase, lane, slot, j)`, the fill word of a scratch slot (`verify::scratch_fill`),
+/// with the program's seed words as literals.
+fn scratch_prelude(p: &Program, dialect: CoreDialect) -> String {
+ if !p.has_scratch() {
+ return String::new();
+ }
+ let (u, fn_) = match dialect {
+ CoreDialect::Metal => ("uint", "inline"),
+ CoreDialect::Cuda => ("uint32_t", "__device__ __forceinline__"),
+ CoreDialect::OpenCl => ("uint", "static inline"),
+ };
+ let mut s = String::new();
+ s.push_str(&format!("// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a {} KiB scratch per warp, {} slots of\n", p.class.scratch_kb, p.class.scratch_slots_per_lane()));
+ s.push_str("// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not\n");
+ s.push_str("// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.\n");
+ s.push_str(&format!(
+ "{fn_} {u} scr_fill({u} gbase, {u} lane, {u} slot, {u} j) {{ {u} sw = (j == 0u) ? {} : ((j == 1u) ? {} : {}); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }}\n",
+ hex(p.seed[0]),
+ hex(p.seed[1]),
+ hex(p.seed[2])
+ ));
+ s
+}
+
+/// Variant 5: one scratch read-modify-write as a statement block. `arena`, `tag`, `gbase` and `lane` are in scope
+/// (the persistent prologue). Reads 16 bytes, folds the three data words into dst, rewrites the slot behind the tag.
+fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str, slot_mask: u32) -> String {
+ let (u, load, store) = match dialect {
+ CoreDialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"),
+ CoreDialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"),
+ CoreDialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena);"),
+ };
+ format!(
+ "{{ {u} s_ = {a} & {slot_mask}u; {load} {u} m_ = (v_.x == tag) ? 0xffffffffu : 0u; {u} w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); {u} w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); {u} w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); {u} x_ = {d} ^ w0_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w1_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w2_; {d} = x_; {store} }}",
+ k = hex(FOLD_MUL)
+ )
+}
+
+/// Variant 5: the persistent-warp prologue. The kernel is launched with N warps (the resident count, the host's
+/// choice); warp `w` owns arena `w` and runs the units `w, w + N, w + 2N, ...` of the launch. Inside the loop the
+/// lottery hash's text is unchanged: `gid` is the unit's first output index plus the lane. The host MUST launch
+/// `groups` as a multiple of N (a uniform trip count: the OpenCL local-memory exchange carries a barrier).
+fn persistent_prologue(dialect: CoreDialect, words_per_lane: usize) -> String {
+ let (a, b) = persistent_prologue_parts(dialect, words_per_lane);
+ a + &b
+}
+
+/// The prologue in two parts: the warp's identity and arena, then the unit loop. OpenCL C requires a `__local`
+/// variable at the outermost scope of the kernel (AMD's compiler enforces it, 5 October 2026, round 3 on the
+/// 9070 XT), so the OpenCL kernels declare the exchange buffer between the two parts.
+fn persistent_prologue_parts(dialect: CoreDialect, words_per_lane: usize) -> (String, String) {
+ let (u, tid, nthreads, ptr) = match dialect {
+ CoreDialect::Metal => ("uint", "tid", "nthreads", "device uint*"),
+ CoreDialect::Cuda => ("uint32_t", "(blockIdx.x * blockDim.x + threadIdx.x)", "(gridDim.x * blockDim.x)", "uint32_t*"),
+ CoreDialect::OpenCl => ("uint", "(uint)get_global_id(0)", "(uint)get_global_size(0)", "__global uint*"),
+ };
+ let mut s = String::new();
+ s.push_str(&format!(" {u} lane = {tid} & 31u;\n"));
+ s.push_str(&format!(" {u} warp_ = {tid} >> 5;\n"));
+ s.push_str(&format!(" {u} nwarps_ = {nthreads} >> 5;\n"));
+ s.push_str(&format!(" {ptr} arena = scratch + ((size_t)warp_ * 32u + lane) * {words_per_lane}u;\n"));
+ let mut l = String::new();
+ l.push_str(&format!(" for ({u} g_ = warp_; g_ < groups; g_ += nwarps_) {{\n"));
+ l.push_str(&format!(" {u} gid = g_ * 32u + lane;\n"));
+ l.push_str(&format!(" {u} gbase = baseNonce + g_ * 32u;\n"));
+ l.push_str(&format!(" {u} tag = salt + g_;\n"));
+ (s, l)
+}
+
+/// The scratch lines of program.h (variant 5).
+fn scratch_header_lines(p: &Program) -> String {
+ if !p.has_scratch() {
+ return String::new();
+ }
+ let mut s = String::new();
+ s.push_str(&format!("// Variant 5: persistent warps, a {} KiB scratch per launched warp (the host launches N warps and passes scratch,\n", p.class.scratch_kb));
+ s.push_str("// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).\n");
+ s.push_str("#define IGNEUM_PERSISTENT_WARPS 1\n");
+ s.push_str(&format!("#define IGNEUM_SCRATCH_OPS {} // scratch read-modify-writes per program ({} per hash)\n", p.class.scratch_slots(), p.scratch_ops_per_hash()));
+ s.push_str(&format!("#define IGNEUM_SCRATCH_SLOTS {}u\n", p.class.scratch_slots_per_lane()));
+ s.push_str(&format!("#define IGNEUM_SCRATCH_WORDS_PER_LANE {}u\n", p.class.scratch_words_per_lane()));
+ s.push_str(&format!("#define IGNEUM_SCRATCH_BYTES_PER_WARP {}u\n", p.class.scratch_bytes_per_warp()));
+ s
+}
+
+/// Hot-table experiment (`docs/plans/hot-table.md`): the `HOT_WORDS` literal of a hot pack's hash kernels (empty
+/// for every other class, so the pinned packs do not change).
+/// One ALU instruction of the latency-shadow block as a statement of the dialect (the same text the program's own
+/// instruction lines use for that op; the block holds no load, scratch, hot or wide op).
+fn shadow_instr_line(dialect: CoreDialect, ins: &Instr) -> String {
+ let d = format!("r{}", ins.dst);
+ let a = format!("r{}", ins.src);
+ let b = format!("r{}", ins.src2);
+ match ins.op {
+ Op::Add => match dialect {
+ CoreDialect::Metal => format!("{d} = {d} + {a} + select({}, {}, ((sel >> {}u) & 1u) != 0u);", hex(ins.imm), hex(ins.imm2), ins.bit),
+ _ => format!("{d} = {d} + {a} + ((((sel >> {}u) & 1u) != 0u) ? {} : {});", ins.bit, hex(ins.imm2), hex(ins.imm)),
+ },
+ Op::Sub => format!("{d} = {d} - {a};"),
+ Op::Mul => format!("{d} = {d} * {a};"),
+ Op::MulHi => match dialect {
+ CoreDialect::Metal => format!("{d} = mulhi({d}, {a});"),
+ CoreDialect::Cuda => format!("{d} = __umulhi({d}, {a});"),
+ CoreDialect::OpenCl => format!("{d} = mul_hi({d}, {a});"),
+ },
+ Op::Xor => format!("{d} = {d} ^ {a};"),
+ Op::Or => format!("{d} = {d} | {a};"),
+ Op::Rotl => format!("{d} = rotl_imm({d}, {}u);", ins.rot),
+ Op::Rotr => format!("{d} = rotr_var({d}, {a});"),
+ Op::Mad => format!("{d} = {a} * {b} + {d};"),
+ Op::Shfl => match dialect {
+ CoreDialect::Metal => format!("{d} = {d} ^ simd_shuffle_xor({a}, (ushort){});", ins.mask),
+ CoreDialect::Cuda => format!("{d} = {d} ^ __shfl_xor_sync(0xffffffffu, {a}, {});", ins.mask),
+ CoreDialect::OpenCl => format!("{{ uint t_; IGNEUM_SHFL_XOR(t_, {a}, {}u); {d} = {d} ^ t_; }}", ins.mask),
+ },
+ Op::Load | Op::WLoad | Op::Scratch | Op::Hot => unreachable!("the shadow block holds ALU instructions only"),
+ }
+}
+
+/// The latency-shadow block (Counter ASIC 3.0 item 8, `docs/analysis/latency-shadow-2026-10-06.md`) inside the
+/// iteration loop, after the program's last instruction: `reps` passes over the block with the iteration's `sel`.
+/// Empty for every program without a shadow, so the pinned v2 and v3 packs do not change by a byte.
+fn shadow_block(p: &Program, dialect: CoreDialect) -> String {
+ if !p.has_shadow() {
+ return String::new();
+ }
+ let reps = p.shadow_reps();
+ let mut s = String::with_capacity(64 * p.shadow.len() + 200);
+ s.push_str(&format!(
+ " // latency-shadow block (Counter ASIC 3.0 item 8): {} ALU instructions x {reps} passes after instruction 63, no load\n",
+ p.shadow.len()
+ ));
+ let ty = if dialect == CoreDialect::Cuda { "uint32_t" } else { "uint" };
+ s.push_str(&format!(" for ({ty} sh = 0u; sh < {reps}u; ++sh) {{\n"));
+ for (k, ins) in p.shadow.iter().enumerate() {
+ s.push_str(&format!(" {} // s{k} {}\n", shadow_instr_line(dialect, ins), ins.op.name()));
+ }
+ s.push_str(" }\n");
+ s
+}
+
+/// The shadow lines of program.h (empty without a shadow).
+fn shadow_header_lines(p: &Program) -> String {
+ let Some(sh) = p.class.shadow else { return String::new() };
+ let mut s = String::new();
+ s.push_str("// Latency-shadow block (Counter ASIC 3.0 item 8, docs/analysis/latency-shadow-2026-10-06.md): NOT the lottery hash. A block of\n");
+ s.push_str("// IGNEUM_SHADOW_INSTRS ALU instructions (the ten non-load families) runs IGNEUM_SHADOW_REPS times at the end of every\n");
+ s.push_str("// iteration; the 16 loads, the acceptance rule and the base program are the class's without the shadow.\n");
+ s.push_str(&format!("#define IGNEUM_SHADOW_INSTRS {}\n", sh.instrs));
+ s.push_str(&format!("#define IGNEUM_SHADOW_REPS {}\n", sh.reps));
+ s.push_str(&format!("#define IGNEUM_SHADOW_INSTRS_PER_HASH {}\n", p.shadow_instrs_per_hash()));
+ s.push_str(&format!("#define IGNEUM_SHADOW_OP_MIX {}\n", jstr(&p.shadow_op_mix())));
+ s
+}
+
+fn hot_define(p: &Program) -> String {
+ match p.class.hot {
+ Some(h) => format!("// Hot table ({} MiB, {} of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md.
+#define HOT_WORDS {}
+", h.mb, h.k, hex(hot_words(h.mb as u32))),
+ None => String::new(),
+ }
+}
+
+/// The hot load as one statement per dialect: the high 32 bits of `src x HOT_WORDS` index the table.
+fn hot_stmt(dialect: CoreDialect, d: &str, a: &str) -> String {
+ match dialect {
+ CoreDialect::Metal => format!("{d} = {d} ^ hot[mulhi({a}, HOT_WORDS)];"),
+ CoreDialect::Cuda => format!("{d} = {d} ^ hot[__umulhi({a}, HOT_WORDS)];"),
+ CoreDialect::OpenCl => format!("{d} = {d} ^ hot[mul_hi({a}, HOT_WORDS)];"),
+ }
+}
+
+/// The hot table's fill core: `ht_segment(hot, seg)`, the cache chain of `mh_cache_segment` under the hot key and
+/// the hot tag (`mh_chacha_block` and `MH_SEGMENT_LINES` must be in scope: the memory-hard core comes first).
+fn emit_hot_core(p: &Program, dialect: CoreDialect) -> String {
+ let Some(h) = p.class.hot else { return String::new() };
+ let (u, fn_, wptr) = match dialect {
+ CoreDialect::Metal => ("uint", "inline", "device uint*"),
+ CoreDialect::Cuda => ("uint32_t", "IGNEUM_HD", "uint32_t*"),
+ CoreDialect::OpenCl => ("uint", "static inline", "__global uint*"),
+ };
+ let k = hot_key(&p.seed_bytes);
+ let mut s = String::with_capacity(1500);
+ s.push_str(&format!(
+ "// Hot table (docs/plans/hot-table.md): {} MiB = {} segments of {} chained ChaCha{} lines under the hot key KH = seed_words(\"igneum-hot/\" || epoch seed bytes), tag \"Igne\" \"umHT\". The cache chain with another key and tag.\n",
+ h.mb,
+ hot_segments(h.mb as u32),
+ CACHE_LINES_PER_SEGMENT,
+ CHACHA_ROUNDS
+ ));
+ s.push_str(&format!("{fn_} void ht_segment({wptr} hot, {u} seg) {{\n"));
+ s.push_str(&format!(" {u} prev[16]; {u} x[16]; {u} y[16];\n"));
+ s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) prev[i] = 0u;\n"));
+ s.push_str(&format!(" for ({u} j = 0u; j < MH_SEGMENT_LINES; ++j) {{\n"));
+ s.push_str(&format!(
+ " x[0] = {} ^ prev[0]; x[1] = {} ^ prev[1]; x[2] = {} ^ prev[2]; x[3] = {} ^ prev[3];\n",
+ hex(CHACHA_SIGMA[0]),
+ hex(CHACHA_SIGMA[1]),
+ hex(CHACHA_SIGMA[2]),
+ hex(CHACHA_SIGMA[3])
+ ));
+ for i in 0..8 {
+ s.push_str(&format!(" x[{}] = {} ^ prev[{}];\n", 4 + i, hex(k[i]), 4 + i));
+ }
+ s.push_str(&format!(
+ " x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = {} ^ prev[14]; x[15] = {} ^ prev[15];\n",
+ hex(HOT_TAG[0]),
+ hex(HOT_TAG[1])
+ ));
+ s.push_str(" mh_chacha_block(x, y);\n");
+ s.push_str(&format!(" {wptr} line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u);\n"));
+ s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) {{ line[i] = y[i]; prev[i] = y[i]; }}\n"));
+ s.push_str(" }\n");
+ s.push_str("}\n");
+ s
+}
+
+/// The hot lines of program.h.
+fn hot_header_lines(p: &Program) -> String {
+ let Some(h) = p.class.hot else { return String::new() };
+ let mut s = String::new();
+ s.push_str("// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the\n");
+ s.push_str("// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under\n");
+ s.push_str("// KH = seed_words(\"igneum-hot/\" || epoch seed bytes) and the tag \"Igne\" \"umHT\"; filled by igneum_hot_fill once per epoch.\n");
+ s.push_str(&format!("#define IGNEUM_HOT_MB {}\n", h.mb));
+ s.push_str(&format!("#define IGNEUM_HOT_WORDS {}\n", hex(hot_words(h.mb as u32))));
+ s.push_str(&format!("#define IGNEUM_HOT_SEGMENTS {}u\n", hot_segments(h.mb as u32)));
+ s.push_str(&format!("#define IGNEUM_HOT_SLOTS {} // hot loads per program ({} per hash), {} the dataset loads ({} of them)\n", h.k, p.hot_loads_per_hash(), if h.added { "added beside" } else { "replacing" }, p.class.dataset_slots()));
+ s.push_str(&format!("#define IGNEUM_HOT_ADDED {}\n", h.added as u8));
+ s.push_str(&format!("#define IGNEUM_HOT_KEY_INIT {{ {} }}\n", join_hex(&hot_key(&p.seed_bytes))));
+ s
+}
+
+pub fn hex(v: u32) -> String {
+ format!("0x{v:08x}u")
+}
+pub fn hex64(v: u64) -> String {
+ format!("0x{v:016x}ull")
+}
+fn jhex(v: u32) -> String {
+ format!("\"0x{v:08x}\"")
+}
+fn jhex64(v: u64) -> String {
+ format!("\"0x{v:016x}\"")
+}
+/// JSON string with the three escapes the Swift applies (quote, backslash, newline).
+fn jstr(s: &str) -> String {
+ let mut o = String::with_capacity(s.len() + 2);
+ o.push('"');
+ for c in s.chars() {
+ match c {
+ '"' => o.push_str("\\\""),
+ '\\' => o.push_str("\\\\"),
+ '\n' => o.push_str("\\n"),
+ _ => o.push(c),
+ }
+ }
+ o.push('"');
+ o
+}
+fn join_hex(v: &[u32]) -> String {
+ v.iter().map(|&x| hex(x)).collect::>().join(", ")
+}
+fn join_jhex(v: &[u32]) -> String {
+ v.iter().map(|&x| jhex(x)).collect::>().join(", ")
+}
+
+/// The three dialects of the memory-hard core.
+#[derive(Clone, Copy, Debug, PartialEq, Eq)]
+pub enum CoreDialect {
+ Metal,
+ Cuda,
+ OpenCl,
+}
+
+/// How the hash kernel gets dataset words. `Stored` is the honest kernel; the inline variants are the
+/// shortcut measurements of MEMHARD.md section 2.2.
+pub enum LoadSource<'a> {
+ Stored,
+ InlineClosed(u32, u32),
+ InlineMemhard(&'a MixParams),
+}
+
+fn log2_segments(shape: &Shape) -> usize {
+ shape.log2_segments() as usize
+}
+
+/// The memory-hard core as source text (`emitMemhardCore`). Every parameter is a literal. Linear layout.
+pub fn emit_memhard_core(mp: &MixParams, dialect: CoreDialect) -> String {
+ emit_memhard_core_layout(mp, dialect, Layout::LINEAR)
+}
+
+/// [`emit_memhard_core`] with the dataset layout: with a non-linear layout `mh_word` and the build kernels go
+/// through `mh_t`, `mh_j` and `mh_addr` (era layout); with the linear layout the text is unchanged.
+pub fn emit_memhard_core_layout(mp: &MixParams, dialect: CoreDialect, layout: Layout) -> String {
+ let shape = &mp.shape;
+ let m = shape.mixer_mult;
+ let cache_log2_words = shape.cache_log2_words;
+ let cache_line_mask = shape.cache_line_mask();
+ let (u, fn_, cptr, wptr, lptr, lcptr) = match dialect {
+ CoreDialect::Metal => {
+ ("uint", "inline", "device const uint*", "device uint*", "thread uint*", "const thread uint*")
+ }
+ CoreDialect::Cuda => ("uint32_t", "IGNEUM_HD", "const uint32_t*", "uint32_t*", "uint32_t*", "const uint32_t*"),
+ CoreDialect::OpenCl => {
+ ("uint", "static inline", "__global const uint*", "__global uint*", "uint*", "const uint*")
+ }
+ };
+ let k = &mp.key;
+ let r = &mp.rot;
+ let mul = &mp.mul;
+ let c = &mp.rc;
+ let mut s = String::with_capacity(6000);
+ s.push_str(&format!(
+ "// Memory-hard dataset core (MEMHARD.md). Cache: 2^{} words in 2^{} segments of {} chained ChaCha{} lines.\n",
+ cache_log2_words,
+ log2_segments(shape),
+ CACHE_LINES_PER_SEGMENT,
+ CHACHA_ROUNDS
+ ));
+ if mp.derive.is_some() {
+ s.push_str("// Item: 8 rounds of (the day's round program + one 64-byte cache read), then round program 8. All parameters are literals.\n");
+ } else if m == 1 {
+ s.push_str("// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.\n");
+ } else {
+ s.push_str(&format!("// Item: 8 rounds of {m} x seed-parameterised mixer + one 64-byte cache read, then {m} x final mixer (class v3, mixer multiplier {m},\n"));
+ s.push_str("// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.\n");
+ }
+ if let Some(dp) = &mp.derive {
+ s.push_str(&format!("// Counter ASIC 3.0 item 2 (docs/plans/counter-asic-3-derivation.md): the mixer slots run the day's drawn program, {} instructions per round\n", dp.len));
+ s.push_str(&format!("// program (mh_round_0..{}), {} per item, drawn from the day key stream after the mixer constants (attempt {}, fingerprint {:016x}).\n", DERIVE_PROGRAMS - 1, dp.instr_count(), dp.attempt, dp.fingerprint()));
+ }
+ s.push_str(&format!("#define MH_CACHE_LINE_MASK {}\n", hex(cache_line_mask)));
+ s.push_str(&format!("#define MH_SEGMENT_LINES {}u\n", CACHE_LINES_PER_SEGMENT));
+ s.push_str("#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }\n");
+ s.push_str(&format!(
+ "{fn_} {u} mh_rotl({u} x, {u} n) {{ return (x << n) | (x >> (32u - n)); }} // n in 1..31 at every call site\n"
+ ));
+ s.push('\n');
+ s.push_str(&format!("// y = ChaCha{CHACHA_ROUNDS} core(x) + x\n"));
+ s.push_str(&format!("{fn_} void mh_chacha_block({lcptr} x, {lptr} y) {{\n"));
+ s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) y[i] = x[i];\n"));
+ s.push_str(&format!(" for ({u} r = 0u; r < {}u; ++r) {{\n", CHACHA_ROUNDS / 2));
+ s.push_str(
+ " MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)\n",
+ );
+ s.push_str(
+ " MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)\n",
+ );
+ s.push_str(
+ " MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)\n",
+ );
+ s.push_str(
+ " MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)\n",
+ );
+ s.push_str(" }\n");
+ s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) y[i] += x[i];\n"));
+ s.push_str("}\n");
+ s.push('\n');
+ s.push_str(&format!(
+ "// One cache segment: {} chained lines written at cache[seg * {}]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.\n",
+ CACHE_LINES_PER_SEGMENT,
+ CACHE_LINES_PER_SEGMENT * 16
+ ));
+ s.push_str(&format!("{fn_} void mh_cache_segment({wptr} cache, {u} seg) {{\n"));
+ s.push_str(&format!(" {u} prev[16]; {u} x[16]; {u} y[16];\n"));
+ s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) prev[i] = 0u;\n"));
+ s.push_str(&format!(" for ({u} j = 0u; j < MH_SEGMENT_LINES; ++j) {{\n"));
+ s.push_str(&format!(
+ " x[0] = {} ^ prev[0]; x[1] = {} ^ prev[1]; x[2] = {} ^ prev[2]; x[3] = {} ^ prev[3];\n",
+ hex(CHACHA_SIGMA[0]),
+ hex(CHACHA_SIGMA[1]),
+ hex(CHACHA_SIGMA[2]),
+ hex(CHACHA_SIGMA[3])
+ ));
+ for i in 0..8 {
+ s.push_str(&format!(" x[{}] = {} ^ prev[{}];\n", 4 + i, hex(k[i]), 4 + i));
+ }
+ s.push_str(&format!(
+ " x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = {} ^ prev[14]; x[15] = {} ^ prev[15];\n",
+ hex(CACHE_TAG[0]),
+ hex(CACHE_TAG[1])
+ ));
+ s.push_str(" mh_chacha_block(x, y);\n");
+ s.push_str(&format!(" {wptr} line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);\n"));
+ s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) {{ line[i] = y[i]; prev[i] = y[i]; }}\n"));
+ s.push_str(" }\n");
+ s.push_str("}\n");
+ s.push('\n');
+ s.push_str("// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.\n");
+ s.push_str(&format!("{fn_} void mh_mixer({lptr} s, {u} rk) {{\n"));
+ for i in 0..16 {
+ s.push_str(&format!(" s[{i}] = (s[{i}] ^ ({} + rk)) * {};\n", hex(c[i]), hex(mul[i])));
+ }
+ let col = (0..4).map(|i| format!("{}u", r[i])).collect::>().join(", ");
+ let dia = (4..8).map(|i| format!("{}u", r[i])).collect::>().join(", ");
+ s.push_str(&format!(" MH_QR(s[0], s[4], s[8], s[12], {col}) MH_QR(s[1], s[5], s[9], s[13], {col})\n"));
+ s.push_str(&format!(" MH_QR(s[2], s[6], s[10], s[14], {col}) MH_QR(s[3], s[7], s[11], s[15], {col})\n"));
+ s.push_str(&format!(" MH_QR(s[0], s[5], s[10], s[15], {dia}) MH_QR(s[1], s[6], s[11], s[12], {dia})\n"));
+ s.push_str(&format!(" MH_QR(s[2], s[7], s[8], s[13], {dia}) MH_QR(s[3], s[4], s[9], s[14], {dia})\n"));
+ s.push_str("}\n");
+ s.push('\n');
+ if m == 1 {
+ s.push_str(&format!(
+ "// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of mixer + cache line s[0] & mask; final mixer.\n"
+ ));
+ } else {
+ s.push_str(&format!(
+ "// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of {m} x mixer + cache line s[0] & mask; {m} x final mixer.\n"
+ ));
+ }
+ if let Some(dp) = &mp.derive {
+ // the nine round programs as straight-line functions over the 16-word state (the same text in every dialect)
+ for (r, prog) in dp.rounds.iter().enumerate() {
+ s.push_str(&format!("// Round program {r}: {} instructions, chain rule (every instruction reads the register the previous one wrote; s[0] first).\n", prog.len()));
+ s.push_str(&format!("{fn_} void mh_round_{r}({lptr} s) {{\n"));
+ for ins in prog {
+ s.push_str(" ");
+ s.push_str(&derive_instr_text(ins));
+ s.push('\n');
+ }
+ s.push_str("}\n");
+ }
+ s.push_str(&format!("// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of (round program r, cache line s[0] & mask); round program {ITEM_ROUNDS}.\n"));
+ s.push_str(&item_signature(shape.state, fn_, cptr, u, lptr));
+ for i in 0..8 {
+ s.push_str(&format!(" s[{i}] = {};\n", hex(k[i])));
+ }
+ for i in 0..8 {
+ s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(mul[i]), hex(c[i])));
+ }
+ if shape.state {
+ s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= leaf[i];\n"));
+ }
+ for r in 0..ITEM_ROUNDS {
+ s.push_str(&format!(" mh_round_{r}(s);\n"));
+ s.push_str(&format!(" {{ {cptr} line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); for ({u} i = 0u; i < 16u; ++i) s[i] ^= line[i]; }}\n"));
+ }
+ s.push_str(&format!(" mh_round_{ITEM_ROUNDS}(s);\n"));
+ s.push_str("}\n");
+ return finish_memhard_core(s, layout, u, fn_, cptr, shape.state);
+ }
+ s.push_str(&item_signature(shape.state, fn_, cptr, u, lptr));
+ for i in 0..8 {
+ s.push_str(&format!(" s[{i}] = {};\n", hex(k[i])));
+ }
+ for i in 0..8 {
+ s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(mul[i]), hex(c[i])));
+ }
+ if shape.state {
+ // class v5: the window's state leaf of item t, before the first mixer (docs/design/class-v5-stored-state.md)
+ s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= leaf[i];\n"));
+ }
+ s.push_str(&format!(" for ({u} r = 0u; r < {ITEM_ROUNDS}u; ++r) {{\n"));
+ if m == 1 {
+ s.push_str(" mh_mixer(s, 0x9E3779B9u * (r + 1u));\n");
+ } else {
+ s.push_str(&format!(" for ({u} j = 0u; j < {m}u; ++j) mh_mixer(s, 0x9E3779B9u * (r * {m}u + j + 1u));\n"));
+ }
+ s.push_str(&format!(" {cptr} line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);\n"));
+ s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= line[i];\n"));
+ s.push_str(" }\n");
+ if m == 1 {
+ s.push_str(&format!(" mh_mixer(s, 0x9E3779B9u * {}u);\n", ITEM_ROUNDS + 1));
+ } else {
+ s.push_str(&format!(
+ " for ({u} j = 0u; j < {m}u; ++j) mh_mixer(s, 0x9E3779B9u * ({}u + j + 1u));\n",
+ ITEM_ROUNDS as u32 * m
+ ));
+ }
+ s.push_str("}\n");
+ finish_memhard_core(s, layout, u, fn_, cptr, shape.state)
+}
+
+/// The `mh_item` signature: under a state shape (class v5) the item takes its 16-word leaf (`leaves + 16 (t mod n)`).
+fn item_signature(state: bool, fn_: &str, cptr: &str, u: &str, lptr: &str) -> String {
+ if state {
+ format!("{fn_} void mh_item({cptr} cache, {cptr} leaf, {u} t, {lptr} s) {{\n")
+ } else {
+ format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n")
+ }
+}
+
+/// The tail of the memhard core: `mh_word` (and the era layout helpers) after `mh_item`. Under a state shape
+/// `mh_word` takes the leaves and their count and derives item t's leaf as `leaves + 16 (t mod nLeaves)`.
+fn finish_memhard_core(mut s: String, layout: Layout, u: &str, fn_: &str, cptr: &str, state: bool) -> String {
+ if state {
+ s.push_str("// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).\n");
+ s.push_str(&format!("{fn_} {cptr} mh_leaf({cptr} leaves, {u} nLeaves, {u} t) {{ return leaves + ((t % nLeaves) * 16u); }}\n"));
+ }
+ if layout.is_linear() {
+ s.push_str("// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.\n");
+ if state {
+ s.push_str(&format!(
+ "{fn_} {u} mh_word({cptr} cache, {cptr} leaves, {u} nLeaves, {u} w) {{ {u} s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, w >> 4u), w >> 4u, s); return s[w & 15u]; }}\n"
+ ));
+ } else {
+ s.push_str(&format!(
+ "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }}\n"
+ ));
+ }
+ } else {
+ s.push_str(&layout_helpers(layout, u, fn_));
+ s.push_str("// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).\n");
+ if state {
+ s.push_str(&format!(
+ "{fn_} {u} mh_word({cptr} cache, {cptr} leaves, {u} nLeaves, {u} w) {{ {u} s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, mh_t(w)), mh_t(w), s); return s[mh_j(w)]; }}\n"
+ ));
+ } else {
+ s.push_str(&format!(
+ "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }}\n"
+ ));
+ }
+ }
+ s
+}
+
+/// The dataset store of one item in a build kernel: `d[i] = s[i]` at `ds + t * 16` for the linear layout, else the
+/// scatter `ds[mh_addr(t, i)] = s[i]` (era layout).
+fn build_store(layout: Layout, dialect: CoreDialect, ds: &str, t: &str) -> String {
+ let (u, wptr, cast) = match dialect {
+ CoreDialect::Metal => ("uint", "device uint*", ""),
+ CoreDialect::Cuda => ("uint32_t", "uint32_t*", "(size_t)"),
+ CoreDialect::OpenCl => ("uint", "__global uint*", "(ulong)"),
+ };
+ if layout.is_linear() {
+ match dialect {
+ CoreDialect::Metal => format!(" {wptr} d = {ds} + {t} * 16u;\n for ({u} i = 0u; i < 16u; ++i) d[i] = s[i];\n"),
+ CoreDialect::Cuda => format!(" {wptr} d = {ds} + (size_t){t} * 16u;\n for ({u} i = 0u; i < 16u; ++i) d[i] = s[i];\n"),
+ CoreDialect::OpenCl => format!(" {wptr} d = {ds} + ((ulong){t} * 16u);\n for ({u} i = 0u; i < 16u; ++i) d[i] = s[i];\n"),
+ }
+ } else {
+ let indent = if dialect == CoreDialect::Metal { " " } else { " " };
+ format!("{indent}for ({u} i = 0u; i < 16u; ++i) {ds}[{cast}mh_addr({t}, i)] = s[i];\n")
+ }
+}
+
+/// [`metal_memhard`] plus, for a hot pack, the hot table's `ht_segment` and `igneum_hot_fill` kernel (one thread per
+/// segment, `IGNEUM_HOT_SEGMENTS` threads). Byte-identical to [`metal_memhard`] for every other class.
+pub fn metal_memhard_for(p: &Program, mp: &MixParams) -> String {
+ let mut s = metal_memhard_layout(mp, p.class.layout());
+ if p.has_hot() {
+ s.push('\n');
+ s.push_str(&emit_hot_core(p, CoreDialect::Metal));
+ s.push_str("// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads).\n");
+ s.push_str("kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) {\n");
+ s.push_str(" ht_segment(hot, gid);\n");
+ s.push_str("}\n");
+ }
+ s
+}
+
+/// Metal library with the cache fill and dataset build kernels for one day key (`memhardMSL`, memhard.metal).
+pub fn metal_memhard(mp: &MixParams) -> String {
+ metal_memhard_layout(mp, Layout::LINEAR)
+}
+
+/// [`metal_memhard`] with the dataset layout (era layout).
+pub fn metal_memhard_layout(mp: &MixParams, layout: Layout) -> String {
+ let mut s = String::new();
+ s.push_str("#include \n");
+ s.push_str("using namespace metal;\n");
+ s.push_str(&emit_memhard_core_layout(mp, CoreDialect::Metal, layout));
+ s.push('\n');
+ s.push_str(&format!("// One thread per segment (2^{} threads).\n", log2_segments(&mp.shape)));
+ s.push_str(
+ "kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {\n",
+ );
+ s.push_str(" mh_cache_segment(cache, gid);\n");
+ s.push_str("}\n");
+ s.push_str("// One thread per 64-byte item (dataset words / 16 threads).\n");
+ if mp.shape.state {
+ s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) in buffer 2, their count in buffer 3.\n");
+ s.push_str(
+ "kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n",
+ );
+ s.push_str(" device const uint* leaves [[buffer(2)]], constant uint& nLeaves [[buffer(3)]],\n");
+ s.push_str(" uint gid [[thread_position_in_grid]]) {\n");
+ s.push_str(" uint s[16];\n");
+ s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, gid), gid, s);\n");
+ } else {
+ s.push_str(
+ "kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n",
+ );
+ s.push_str(" uint gid [[thread_position_in_grid]]) {\n");
+ s.push_str(" uint s[16];\n");
+ s.push_str(" mh_item(cache, gid, s);\n");
+ }
+ s.push_str(&build_store(layout, CoreDialect::Metal, "dataset", "gid"));
+ s.push_str("}\n");
+ s
+}
+
+const DS_ELEM_BODY: &str = " x *= 0x9E3779B1u; x ^= x >> 15;\n x += d1;\n x *= 0x85EBCA77u; x ^= x >> 13;\n x *= 0xC2B2AE3Du; x ^= x >> 16;\n return x;\n}\n";
+
+/// The Metal hash kernel (`generateMSL`, program.metal).
+pub fn metal_program(p: &Program, dataset_log2: u32, source: LoadSource) -> String {
+ metal_program_impl(p, dataset_log2, source, false)
+}
+
+/// The header-bound Metal kernel (`program_bound.metal`, serve mode of proto-metal): `igneum_hash_bound` reads its
+/// init words `I` from `constant uint* initw [[buffer(3)]]` (`bind::block_init_words`) instead of `SEEDW`. Same
+/// instruction text as `igneum_hash`. Stored dataset only.
+pub fn metal_program_bound(p: &Program, dataset_log2: u32) -> String {
+ metal_program_impl(p, dataset_log2, LoadSource::Stored, true)
+}
+
+fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: bool) -> String {
+ let mask = mask_for(dataset_log2);
+ let mut s = String::with_capacity(5000);
+ s.push_str("#include \n");
+ s.push_str("using namespace metal;\n");
+ s.push('\n');
+ s.push_str(&format!("#define MASK {}\n", hex(mask)));
+ s.push_str(&hot_define(p));
+ s.push_str(&format!("constant uint SEEDW[8] = {{ {} }};\n", join_hex(&p.seed)));
+ s.push('\n');
+ s.push_str("inline uint splitmix32(uint x) {\n");
+ s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n");
+ s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n");
+ s.push_str(" x ^= x >> 16;\n");
+ s.push_str(" return x;\n");
+ s.push_str("}\n");
+ s.push_str("inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31\n");
+ s.push_str("inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n");
+ s.push_str("inline uint ds_elem(uint i, uint d0, uint d1) {\n");
+ s.push_str(" uint x = i ^ d0;\n");
+ s.push_str(DS_ELEM_BODY);
+ s.push('\n');
+ if p.has_wide() {
+ s.push_str("#define WMASK (MASK & ~31u)\n\n");
+ }
+ let mut buffer0 = "device const uint* dataset [[buffer(0)]]";
+ if let LoadSource::InlineMemhard(mp) = &source {
+ s.push_str(&emit_memhard_core_layout(mp, CoreDialect::Metal, p.class.layout()));
+ s.push('\n');
+ buffer0 = "device const uint* cache [[buffer(0)]]";
+ }
+ s.push_str(&scratch_prelude(p, CoreDialect::Metal));
+ if bound {
+ s.push_str("// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.\n");
+ s.push_str(&format!("kernel void igneum_hash_bound({buffer0},\n"));
+ } else {
+ s.push_str(&format!("kernel void igneum_hash({buffer0},\n"));
+ }
+ s.push_str(" device ulong* out [[buffer(1)]],\n");
+ s.push_str(" constant uint& baseNonce [[buffer(2)]],\n");
+ if bound {
+ s.push_str(" constant uint* initw [[buffer(3)]],\n");
+ }
+ if p.has_hot() {
+ s.push_str(&format!(" device const uint* hot [[buffer({})]],\n", if bound { 4 } else { 3 }));
+ }
+ if p.has_scratch() {
+ let b = (if bound { 4 } else { 3 }) + p.has_hot() as usize;
+ s.push_str(&format!(" device uint* scratch [[buffer({b})]],\n"));
+ s.push_str(&format!(" constant uint& groups [[buffer({})]],\n", b + 1));
+ s.push_str(&format!(" constant uint& salt [[buffer({})]],\n", b + 2));
+ s.push_str(" uint tid [[thread_position_in_grid]],\n");
+ s.push_str(" uint nthreads [[threads_per_grid]]) {\n");
+ s.push_str(&persistent_prologue(CoreDialect::Metal, p.class.scratch_words_per_lane()));
+ } else {
+ s.push_str(" uint gid [[thread_position_in_grid]]) {\n");
+ }
+ s.push_str(" uint nonce = baseNonce + gid;\n");
+ s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n");
+ if p.has_wide() {
+ s.push_str(" uint lane = gid & 31u;\n");
+ }
+ let iw = if bound { "initw" } else { "SEEDW" };
+ for i in 0..8 {
+ s.push_str(&format!(
+ " {{ uint x = nonce ^ {iw}[{i}]; x += 0x9e3779b9u * {}u; x = splitmix32(x); r{i} = x ^ {iw}[{}]; }}\n",
+ i + 1,
+ (i + 1) & 7
+ ));
+ }
+ s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n"));
+ let era = p.class.era;
+ let word_index = |a: &str, wide: bool, ins: &Instr| -> String {
+ if wide {
+ format!("(simd_broadcast({a}, 0) & WMASK) + lane")
+ } else {
+ load_index_expr(CoreDialect::Metal, era.as_ref(), ins, a, dataset_log2)
+ }
+ };
+ let fetch = |idx: String| -> String {
+ match &source {
+ LoadSource::Stored => format!("dataset[{idx}]"),
+ LoadSource::InlineClosed(d0, d1) => format!("ds_elem({idx}, {}, {})", hex(*d0), hex(*d1)),
+ LoadSource::InlineMemhard(_) => format!("mh_word(cache, {idx})"),
+ }
+ };
+ for (k, ins) in p.instrs.iter().enumerate() {
+ let d = format!("r{}", ins.dst);
+ let a = format!("r{}", ins.src);
+ let b = format!("r{}", ins.src2);
+ let line = match ins.op {
+ Op::Add => format!(
+ "{d} = {d} + {a} + select({}, {}, ((sel >> {}u) & 1u) != 0u);",
+ hex(ins.imm),
+ hex(ins.imm2),
+ ins.bit
+ ),
+ Op::Sub => format!("{d} = {d} - {a};"),
+ Op::Mul => format!("{d} = {d} * {a};"),
+ Op::MulHi => format!("{d} = mulhi({d}, {a});"),
+ Op::Xor => format!("{d} = {d} ^ {a};"),
+ Op::Or => format!("{d} = {d} | {a};"),
+ Op::Rotl => format!("{d} = rotl_imm({d}, {}u);", ins.rot),
+ Op::Rotr => format!("{d} = rotr_var({d}, {a});"),
+ Op::Mad => format!("{d} = {a} * {b} + {d};"),
+ Op::Shfl => format!("{d} = {d} ^ simd_shuffle_xor({a}, (ushort){});", ins.mask),
+ Op::Load if load_width(ins) > 1 => {
+ let (src, closed) = match &source {
+ LoadSource::Stored => (WideSource::Stored, None),
+ LoadSource::InlineClosed(d0, d1) => (WideSource::InlineClosed, Some((*d0, *d1))),
+ LoadSource::InlineMemhard(_) => (WideSource::InlineMemhard, None),
+ };
+ wide_load_stmt(CoreDialect::Metal, &d, &word_index(&a, false, ins), ins.width, src, closed)
+ }
+ Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&a, false, ins))),
+ Op::WLoad => format!("{d} = {d} ^ {};", fetch(word_index(&a, true, ins))),
+ Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a, p.class.scratch_slot_mask()),
+ Op::Hot => hot_stmt(CoreDialect::Metal, &d, &a),
+ };
+ s.push_str(&format!(" {line} // {k}\n"));
+ }
+ s.push_str(&shadow_block(p, CoreDialect::Metal));
+ s.push_str(" }\n");
+ s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
+ s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n");
+ s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n");
+ if p.has_scratch() {
+ s.push_str(" }\n");
+ }
+ s.push_str("}\n");
+ s
+}
+
+/// The Metal closed-form fill kernel (`fillMSL`).
+pub const METAL_FILL: &str = "#include \nusing namespace metal;\ninline uint ds_elem(uint i, uint d0, uint d1) {\n uint x = i ^ d0;\n x *= 0x9E3779B1u; x ^= x >> 15;\n x += d1;\n x *= 0x85EBCA77u; x ^= x >> 13;\n x *= 0xC2B2AE3Du; x ^= x >> 16;\n return x;\n}\nkernel void igneum_fill(device uint* dataset [[buffer(0)]],\n constant uint2& day [[buffer(1)]],\n uint gid [[thread_position_in_grid]]) {\n dataset[gid] = ds_elem(gid, day.x, day.y);\n}";
+
+fn generated_by(seed: &str) -> String {
+ format!("// Generated by igneum-pow export (generator v{GENERATOR_VERSION}) for seed \"{seed}\". Do not edit by hand.\n")
+}
+
+pub fn hex_bytes(b: &[u8]) -> String {
+ b.iter().map(|x| format!("{x:02x}")).collect()
+}
+
+fn init_line(p: &Program, u: &str, i: usize) -> String {
+ let addc = 0x9e3779b9u32.wrapping_mul(i as u32 + 1);
+ format!(
+ " {{ {u} x = nonce ^ {}; x += {}; x = splitmix32(x); r{i} = x ^ {}; }} // SEEDW[{i}], 0x9e3779b9u * {}u, SEEDW[{}]\n",
+ hex(p.seed[i]),
+ hex(addc),
+ hex(p.seed[(i + 1) & 7]),
+ i + 1,
+ (i + 1) & 7
+ )
+}
+
+/// The instruction lines of the CUDA hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`).
+fn cuda_instr_lines(p: &Program, dataset_log2: u32) -> String {
+ let mut s = String::with_capacity(6000);
+ let era = p.class.era;
+ for (k, ins) in p.instrs.iter().enumerate() {
+ let d = format!("r{}", ins.dst);
+ let a = format!("r{}", ins.src);
+ let b = format!("r{}", ins.src2);
+ let line = match ins.op {
+ // Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well.
+ Op::Add => format!(
+ "{d} = {d} + {a} + ((((sel >> {}u) & 1u) != 0u) ? {} : {});",
+ ins.bit,
+ hex(ins.imm2),
+ hex(ins.imm)
+ ),
+ Op::Sub => format!("{d} = {d} - {a};"),
+ Op::Mul => format!("{d} = {d} * {a};"),
+ Op::MulHi => format!("{d} = __umulhi({d}, {a});"),
+ Op::Xor => format!("{d} = {d} ^ {a};"),
+ Op::Or => format!("{d} = {d} | {a};"),
+ Op::Rotl => format!("{d} = rotl_imm({d}, {}u);", ins.rot),
+ Op::Rotr => format!("{d} = rotr_var({d}, {a});"),
+ Op::Mad => format!("{d} = {a} * {b} + {d};"),
+ Op::Shfl => format!("{d} = {d} ^ __shfl_xor_sync(0xffffffffu, {a}, {});", ins.mask),
+ Op::Load if load_width(ins) > 1 => {
+ wide_load_stmt(CoreDialect::Cuda, &d, &load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None)
+ }
+ Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2)),
+ Op::WLoad => format!("{d} = {d} ^ ds[(__shfl_sync(0xffffffffu, {a}, 0) & wmask) + lane];"),
+ Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a, p.class.scratch_slot_mask()),
+ Op::Hot => hot_stmt(CoreDialect::Cuda, &d, &a),
+ };
+ s.push_str(&format!(" {line} // {k} {}\n", ins.op.name()));
+ }
+ s
+}
+
+/// The CUDA kernel (`generateCUDA`, kernel.cu). `memhard` is `None` for a closed-form pack.
+pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String {
+ cuda_kernel_at(p, memhard, DEFAULT_DATASET_LOG2)
+}
+
+/// [`cuda_kernel`] at a dataset size (an era program's window constants are literals of the pack's size; every
+/// other class ignores it).
+pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String {
+ let layout = p.class.layout();
+ let mut s = String::with_capacity(9000);
+ s.push_str(&generated_by(&p.seed_string));
+ s.push_str(
+ "// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).\n",
+ );
+ s.push_str("// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.\n");
+ s.push_str("#include \n");
+ s.push_str("#include \n");
+ s.push_str("#include \"program.h\"\n");
+ if memhard.is_some() {
+ s.push_str("#include \"memhard.h\"\n");
+ }
+ s.push('\n');
+ s.push_str(&hot_define(p));
+ s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n");
+ s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n");
+ s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n");
+ s.push_str(" x ^= x >> 16;\n");
+ s.push_str(" return x;\n");
+ s.push_str("}\n");
+ s.push_str("// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.\n");
+ s.push_str("__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }\n");
+ s.push_str("// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.\n");
+ s.push_str("__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n");
+ s.push_str("__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {\n");
+ s.push_str(" uint32_t x = i ^ d0;\n");
+ s.push_str(DS_ELEM_BODY);
+ s.push('\n');
+ if memhard.is_none() {
+ s.push_str("// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.\n");
+ s.push_str("__global__ void igneum_fill(uint32_t* ds, uint32_t n, uint32_t d0, uint32_t d1) {\n");
+ s.push_str(" uint32_t i = blockIdx.x * blockDim.x + threadIdx.x;\n");
+ s.push_str(" if (i < n) ds[i] = ds_elem(i, d0, d1);\n");
+ s.push_str("}\n");
+ s.push('\n');
+ } else {
+ s.push_str(
+ "// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.\n",
+ );
+ s.push_str(
+ "// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.\n",
+ );
+ s.push_str("__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {\n");
+ s.push_str(" uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;\n");
+ s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n");
+ s.push_str("}\n");
+ if p.class.state {
+ s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.\n");
+ s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {\n");
+ } else {
+ s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n");
+ }
+ s.push_str(" uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;\n");
+ s.push_str(" if (t < nItems) {\n");
+ s.push_str(" uint32_t s[16];\n");
+ if p.class.state {
+ s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);\n");
+ } else {
+ s.push_str(" mh_item(cache, t, s);\n");
+ }
+ s.push_str(&build_store(layout, CoreDialect::Cuda, "ds", "t"));
+ s.push_str(" }\n");
+ s.push_str("}\n");
+ if p.has_hot() {
+ s.push_str("// Hot table (ht_segment is in memhard.h): one thread per segment.\n");
+ s.push_str("__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) {\n");
+ s.push_str(" uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;\n");
+ s.push_str(" if (seg < nSegments) ht_segment(hot, seg);\n");
+ s.push_str("}\n");
+ }
+ s.push('\n');
+ }
+ if memhard.is_none() && p.has_hot() {
+ panic!("a hot-table pack needs the memory-hard dataset (the hot fill shares its ChaCha core)");
+ }
+ s.push_str("// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every\n");
+ s.push_str("// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a\n");
+ s.push_str("// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.\n");
+ s.push_str(&scratch_prelude(p, CoreDialect::Cuda));
+ let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" };
+ let hot_args = if p.has_hot() { ", const uint32_t* hot" } else { "" };
+ s.push_str(&format!("__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask{hot_args}{scratch_args}) {{\n"));
+ if p.has_scratch() {
+ s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane()));
+ } else {
+ s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n");
+ }
+ s.push_str(" uint32_t nonce = baseNonce + gid;\n");
+ s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n");
+ if p.has_wide() {
+ s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n");
+ }
+ for i in 0..8 {
+ s.push_str(&init_line(p, "uint32_t", i));
+ }
+ s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n"));
+ s.push_str(&cuda_instr_lines(p, dataset_log2));
+ s.push_str(&shadow_block(p, CoreDialect::Cuda));
+ s.push_str(" }\n");
+ s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
+ s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n");
+ s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n");
+ if p.has_scratch() {
+ s.push_str(" }\n");
+ }
+ s.push_str("}\n");
+ s.push('\n');
+ s.push_str("// Host-side launch wrappers. Declared in program.h, called from host.cu.\n");
+ if memhard.is_none() {
+ s.push_str("cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1) {\n");
+ s.push_str(" if (nWords == 0u) return cudaErrorInvalidValue;\n");
+ s.push_str(" uint32_t block = 256u;\n");
+ s.push_str(" uint32_t grid = (nWords + block - 1u) / block;\n");
+ s.push_str(" igneum_fill<<>>(ds, nWords, d0, d1);\n");
+ s.push_str(" return cudaGetLastError();\n");
+ s.push_str("}\n");
+ s.push('\n');
+ } else {
+ s.push_str("cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {\n");
+ s.push_str(" if (nSegments == 0u) return cudaErrorInvalidValue;\n");
+ s.push_str(" uint32_t block = 256u;\n");
+ s.push_str(" uint32_t grid = (nSegments + block - 1u) / block;\n");
+ s.push_str(" igneum_cache_fill<<>>(cache, nSegments);\n");
+ s.push_str(" return cudaGetLastError();\n");
+ s.push_str("}\n");
+ s.push('\n');
+ if p.class.state {
+ s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {\n");
+ s.push_str(" if (nItems == 0u || nLeaves == 0u) return cudaErrorInvalidValue;\n");
+ } else {
+ s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n");
+ s.push_str(" if (nItems == 0u) return cudaErrorInvalidValue;\n");
+ }
+ s.push_str(" uint32_t block = 256u;\n");
+ s.push_str(" uint32_t grid = (nItems + block - 1u) / block;\n");
+ if p.class.state {
+ s.push_str(" igneum_build<<>>(ds, cache, leaves, nLeaves, nItems);\n");
+ } else {
+ s.push_str(" igneum_build<<>>(ds, cache, nItems);\n");
+ }
+ s.push_str(" return cudaGetLastError();\n");
+ s.push_str("}\n");
+ s.push('\n');
+ if p.has_hot() {
+ s.push_str("cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) {\n");
+ s.push_str(" if (nSegments == 0u) return cudaErrorInvalidValue;\n");
+ s.push_str(" uint32_t block = 256u;\n");
+ s.push_str(" uint32_t grid = (nSegments + block - 1u) / block;\n");
+ s.push_str(" igneum_hot_fill<<>>(hot, nSegments);\n");
+ s.push_str(" return cudaGetLastError();\n");
+ s.push_str("}\n");
+ s.push('\n');
+ }
+ }
+ let (hot_decl, hot_pass) = if p.has_hot() { (" const uint32_t* hot,", " hot,") } else { ("", "") };
+ if p.has_scratch() {
+ s.push_str("// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).\n");
+ s.push_str(&format!("cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n"));
+ s.push_str(" uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {\n");
+ s.push_str(" if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;\n");
+ s.push_str(" uint32_t block = 32u * blockWarps;\n");
+ s.push_str(" if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;\n");
+ s.push_str(&format!(" igneum_hash<<>>(ds, out, baseNonce, mask,{hot_pass} scratch, nonces / 32u, salt);\n"));
+ s.push_str(" return cudaGetLastError();\n");
+ s.push_str("}\n");
+ } else {
+ s.push_str(&format!(
+ "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n",
+ ));
+ s.push_str(" uint32_t nonces, uint32_t blockWarps) {\n");
+ s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n");
+ s.push_str(" uint32_t block = 32u * blockWarps;\n");
+ s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n");
+ s.push_str(&format!(" igneum_hash<<>>(ds, out, baseNonce, mask{});\n", if p.has_hot() { ", hot" } else { "" }));
+ s.push_str(" return cudaGetLastError();\n");
+ s.push_str("}\n");
+ }
+ s.push('\n');
+ s.push_str("cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {\n");
+ s.push_str(" cudaFuncAttributes attr;\n");
+ s.push_str(" cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);\n");
+ s.push_str(" if (e != cudaSuccess) return e;\n");
+ s.push_str(" *numRegs = attr.numRegs;\n");
+ s.push_str(" return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);\n");
+ s.push_str("}\n");
+ s
+}
+
+/// `kernel_bound.cu`: the header-bound CUDA hash kernel for the serve mode of proto-cuda. A standalone
+/// translation unit (compiled next to kernel.cu, which keeps the cache-fill and build wrappers): the init words
+/// `I` arrive by value in `IgneumInitWords` (`bind::block_init_words`), the instruction text is that of
+/// `igneum_hash`. Declarations for the host are at the top of the file.
+pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String {
+ cuda_kernel_bound_at(p, memhard, DEFAULT_DATASET_LOG2)
+}
+
+/// [`cuda_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]).
+pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String {
+ let mut s = String::with_capacity(9000);
+ s.push_str(&generated_by(&p.seed_string));
+ s.push_str(
+ "// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.\n",
+ );
+ s.push_str("// Host declarations (also in program_bound.h if present):\n");
+ s.push_str("// struct IgneumInitWords { uint32_t w[8]; };\n");
+ s.push_str("// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n");
+ s.push_str(
+ "// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);\n",
+ );
+ s.push_str("// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);\n");
+ s.push_str("#include \n");
+ s.push_str("#include \n");
+ s.push_str("#include \"program.h\"\n");
+ s.push('\n');
+ s.push_str("struct IgneumInitWords { uint32_t w[8]; };\n");
+ s.push('\n');
+ s.push_str(&hot_define(p));
+ s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n");
+ s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n");
+ s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n");
+ s.push_str(" x ^= x >> 16;\n");
+ s.push_str(" return x;\n");
+ s.push_str("}\n");
+ s.push_str("__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }\n");
+ s.push_str("__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n");
+ s.push('\n');
+ let _ = memhard; // the bound kernel reads the stored dataset in both constructions
+ s.push_str(&scratch_prelude(p, CoreDialect::Cuda));
+ let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" };
+ let hot_args = if p.has_hot() { ", const uint32_t* hot" } else { "" };
+ s.push_str(&format!("__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw{hot_args}{scratch_args}) {{\n"));
+ if p.has_scratch() {
+ s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane()));
+ } else {
+ s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n");
+ }
+ s.push_str(" uint32_t nonce = baseNonce + gid;\n");
+ s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n");
+ if p.has_wide() {
+ s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n");
+ }
+ for i in 0..8 {
+ s.push_str(&format!(
+ " {{ uint32_t x = nonce ^ iw.w[{i}]; x += 0x9e3779b9u * {}u; x = splitmix32(x); r{i} = x ^ iw.w[{}]; }}\n",
+ i + 1,
+ (i + 1) & 7
+ ));
+ }
+ s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n"));
+ s.push_str(&cuda_instr_lines(p, dataset_log2));
+ s.push_str(&shadow_block(p, CoreDialect::Cuda));
+ s.push_str(" }\n");
+ s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
+ s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n");
+ s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n");
+ if p.has_scratch() {
+ s.push_str(" }\n");
+ }
+ s.push_str("}\n");
+ s.push('\n');
+ let (hot_decl, hot_pass) = if p.has_hot() { (" const uint32_t* hot,", ", hot") } else { ("", "") };
+ if p.has_scratch() {
+ s.push_str("// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).\n");
+ s.push_str("cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n");
+ s.push_str(&format!(" IgneumInitWords iw,{hot_decl} uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {{\n"));
+ s.push_str(" if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;\n");
+ s.push_str(" uint32_t block = 32u * blockWarps;\n");
+ s.push_str(" if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;\n");
+ s.push_str(&format!(" igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw{hot_pass}, scratch, nonces / 32u, salt);\n"));
+ s.push_str(" return cudaGetLastError();\n");
+ s.push_str("}\n");
+ } else {
+ s.push_str(
+ "cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n",
+ );
+ s.push_str(&format!(" IgneumInitWords iw,{hot_decl} uint32_t nonces, uint32_t blockWarps) {{\n"));
+ s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n");
+ s.push_str(" uint32_t block = 32u * blockWarps;\n");
+ s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n");
+ s.push_str(&format!(" igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw{hot_pass});\n"));
+ s.push_str(" return cudaGetLastError();\n");
+ s.push_str("}\n");
+ }
+ s.push('\n');
+ s.push_str("cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {\n");
+ s.push_str(" cudaFuncAttributes attr;\n");
+ s.push_str(" cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);\n");
+ s.push_str(" if (e != cudaSuccess) return e;\n");
+ s.push_str(" *numRegs = attr.numRegs;\n");
+ s.push_str(" return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);\n");
+ s.push_str("}\n");
+ s
+}
+
+/// The instruction lines of the OpenCL hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`).
+fn opencl_instr_lines(p: &Program, dataset_log2: u32) -> String {
+ let mut s = String::with_capacity(6000);
+ let era = p.class.era;
+ for (k, ins) in p.instrs.iter().enumerate() {
+ let d = format!("r{}", ins.dst);
+ let a = format!("r{}", ins.src);
+ let b = format!("r{}", ins.src2);
+ let line = match ins.op {
+ Op::Add => format!(
+ "{d} = {d} + {a} + ((((sel >> {}u) & 1u) != 0u) ? {} : {});",
+ ins.bit,
+ hex(ins.imm2),
+ hex(ins.imm)
+ ),
+ Op::Sub => format!("{d} = {d} - {a};"),
+ Op::Mul => format!("{d} = {d} * {a};"),
+ Op::MulHi => format!("{d} = mul_hi({d}, {a});"),
+ Op::Xor => format!("{d} = {d} ^ {a};"),
+ Op::Or => format!("{d} = {d} | {a};"),
+ Op::Rotl => format!("{d} = rotl_imm({d}, {}u);", ins.rot),
+ Op::Rotr => format!("{d} = rotr_var({d}, {a});"),
+ Op::Mad => format!("{d} = {a} * {b} + {d};"),
+ Op::Shfl => format!("{{ uint t_; IGNEUM_SHFL_XOR(t_, {a}, {}u); {d} = {d} ^ t_; }}", ins.mask),
+ Op::Load if load_width(ins) > 1 => {
+ wide_load_stmt(CoreDialect::OpenCl, &d, &load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None)
+ }
+ Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2)),
+ Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[(t_ & wmask) + lane]; }}"),
+ Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a, p.class.scratch_slot_mask()),
+ Op::Hot => hot_stmt(CoreDialect::OpenCl, &d, &a),
+ };
+ s.push_str(&format!(" {line} // {k} {}\n", ins.op.name()));
+ }
+ s
+}
+
+/// `kernel_bound.cl`: `kernel.cl` plus the header-bound kernel `igneum_hash_bound`, whose init words come from a
+/// fifth argument (`__global const uint* initw`, 8 words, `bind::block_init_words`). One source file so the serve
+/// mode of proto-opencl/host.c builds cache fill, dataset build and the bound hash from it at runtime.
+pub fn opencl_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String {
+ opencl_kernel_bound_at(p, memhard, DEFAULT_DATASET_LOG2)
+}
+
+/// [`opencl_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]).
+pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String {
+ let mut s = opencl_kernel_at(p, memhard, dataset_log2);
+ s.push('\n');
+ s.push_str(
+ "// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.\n",
+ );
+ let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" };
+ let hot_args = if p.has_hot() { ", __global const uint* hot" } else { "" };
+ s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw{hot_args}{scratch_args}) {{\n"));
+ let (setup, unit_loop) = persistent_prologue_parts(CoreDialect::OpenCl, p.class.scratch_words_per_lane());
+ if p.has_scratch() {
+ s.push_str(&setup);
+ } else {
+ s.push_str(" uint gid = (uint)get_global_id(0);\n");
+ }
+ s.push_str(" uint lid = (uint)get_local_id(0);\n");
+ if p.has_scratch() {
+ // the __local exchange buffer must sit at the kernel's outermost scope: declare it, then open the unit loop
+ s.push_str("#if IGNEUM_EXCHANGE == 0\n");
+ s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n");
+ s.push_str(" uint xk = 0u;\n");
+ s.push_str("#else\n");
+ s.push_str(" (void)lid;\n");
+ s.push_str("#endif\n");
+ s.push_str(&unit_loop);
+ s.push_str(" uint nonce = baseNonce + gid;\n");
+ s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n");
+ s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n");
+ } else {
+ s.push_str(" uint nonce = baseNonce + gid;\n");
+ s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n");
+ s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n");
+ s.push_str("#if IGNEUM_EXCHANGE == 0\n");
+ s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n");
+ s.push_str(" uint xk = 0u;\n");
+ s.push_str("#else\n");
+ s.push_str(" (void)lid;\n");
+ s.push_str("#endif\n");
+ }
+ if p.has_wide() {
+ s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n");
+ }
+ for i in 0..8 {
+ s.push_str(&format!(
+ " {{ uint x = nonce ^ iw{i}; x += 0x9e3779b9u * {}u; x = splitmix32(x); r{i} = x ^ iw{}; }}\n",
+ i + 1,
+ (i + 1) & 7
+ ));
+ }
+ s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n"));
+ s.push_str(&opencl_instr_lines(p, dataset_log2));
+ s.push_str(&shadow_block(p, CoreDialect::OpenCl));
+ s.push_str(" }\n");
+ s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
+ s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n");
+ s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n");
+ if p.has_scratch() {
+ s.push_str(" }\n");
+ }
+ s.push_str("}\n");
+ s
+}
+
+/// The OpenCL C 1.2 kernel (`generateOpenCL`, kernel.cl).
+pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String {
+ opencl_kernel_at(p, memhard, DEFAULT_DATASET_LOG2)
+}
+
+/// [`opencl_kernel`] at a dataset size (see [`cuda_kernel_at`]).
+pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String {
+ let layout = p.class.layout();
+ let mut s = String::with_capacity(14000);
+ s.push_str(&generated_by(&p.seed_string));
+ s.push_str("// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).\n");
+ s.push_str("// Built from source at runtime by proto-opencl/host.c, which passes these defines:\n");
+ s.push_str("// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)\n");
+ s.push_str(
+ "// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)\n",
+ );
+ s.push_str("// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32\n");
+ s.push_str("// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition\n");
+ s.push_str("// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;\n");
+ s.push_str("// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.\n");
+ s.push_str("#ifndef IGNEUM_GROUP\n#define IGNEUM_GROUP 32\n#endif\n");
+ s.push_str("#ifndef IGNEUM_EXCHANGE\n#define IGNEUM_EXCHANGE 0\n#endif\n");
+ s.push_str("#ifdef __OPENCL_VERSION__\n");
+ s.push_str("#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))\n");
+ s.push_str("#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]\n");
+ s.push_str("#if IGNEUM_EXCHANGE == 1\n");
+ s.push_str("#ifdef cl_khr_subgroups\n#pragma OPENCL EXTENSION cl_khr_subgroups : enable\n#endif\n");
+ s.push_str("#ifdef cl_khr_subgroup_shuffle\n#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable\n#endif\n");
+ s.push_str("#elif IGNEUM_EXCHANGE == 2\n#pragma OPENCL EXTENSION cl_intel_subgroups : enable\n#endif\n");
+ if p.has_scratch() {
+ s.push_str("#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))\n");
+ }
+ s.push_str("#else\n");
+ s.push_str("// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.\n");
+ s.push_str("#include \"emu_opencl.h\"\n#endif\n");
+ s.push('\n');
+ s.push_str("#if IGNEUM_EXCHANGE == 1\n");
+ s.push_str("#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))\n");
+ s.push_str("#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)\n");
+ s.push_str("#elif IGNEUM_EXCHANGE == 2\n");
+ s.push_str("#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))\n");
+ s.push_str("#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)\n");
+ s.push_str("#else\n");
+ s.push_str("// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per\n");
+ s.push_str("// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane\n");
+ s.push_str("// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's\n");
+ s.push_str(
+ "// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.\n",
+ );
+ s.push_str("#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }\n");
+ s.push_str("#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }\n");
+ s.push_str("#endif\n");
+ s.push('\n');
+ s.push_str(&hot_define(p));
+ s.push_str("static inline uint splitmix32(uint x) {\n");
+ s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n");
+ s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n");
+ s.push_str(" x ^= x >> 16;\n");
+ s.push_str(" return x;\n");
+ s.push_str("}\n");
+ s.push_str("// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.\n");
+ s.push_str("static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }\n");
+ s.push_str("// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.\n");
+ s.push_str("static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }\n");
+ s.push_str("static inline uint ds_elem(uint i, uint d0, uint d1) {\n");
+ s.push_str(" uint x = i ^ d0;\n");
+ s.push_str(DS_ELEM_BODY);
+ s.push('\n');
+ if let Some(mp) = memhard {
+ s.push_str(&emit_memhard_core_layout(mp, CoreDialect::OpenCl, layout));
+ s.push('\n');
+ s.push_str("// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.\n");
+ s.push_str("// The same constants as memhard.h in this pack (one emitter, three dialects).\n");
+ s.push_str("__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {\n");
+ s.push_str(" uint seg = (uint)get_global_id(0);\n");
+ s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n");
+ s.push_str("}\n");
+ if p.class.state {
+ s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.\n");
+ s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, __global const uint* leaves, uint nLeaves, uint nItems) {\n");
+ } else {
+ s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {\n");
+ }
+ s.push_str(" uint t = (uint)get_global_id(0);\n");
+ s.push_str(" if (t < nItems) {\n");
+ s.push_str(" uint s[16];\n");
+ if p.class.state {
+ s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);\n");
+ } else {
+ s.push_str(" mh_item(cache, t, s);\n");
+ }
+ s.push_str(&build_store(layout, CoreDialect::OpenCl, "ds", "t"));
+ s.push_str(" }\n");
+ s.push_str("}\n");
+ if p.has_hot() {
+ s.push_str(&emit_hot_core(p, CoreDialect::OpenCl));
+ s.push_str("// Hot table: one work-item per segment.\n");
+ s.push_str("__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) {\n");
+ s.push_str(" uint seg = (uint)get_global_id(0);\n");
+ s.push_str(" if (seg < nSegments) ht_segment(hot, seg);\n");
+ s.push_str("}\n");
+ }
+ s.push('\n');
+ } else if p.has_hot() {
+ panic!("a hot-table pack needs the memory-hard dataset (the hot fill shares its ChaCha core)");
+ } else {
+ s.push_str("// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.\n");
+ s.push_str("__kernel void igneum_fill(__global uint* ds, uint n, uint d0, uint d1) {\n");
+ s.push_str(" uint i = (uint)get_global_id(0);\n");
+ s.push_str(" if (i < n) ds[i] = ds_elem(i, d0, d1);\n");
+ s.push_str("}\n");
+ s.push('\n');
+ }
+ s.push_str("// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the\n");
+ s.push_str("// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and\n");
+ s.push_str("// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).\n");
+ s.push_str(&scratch_prelude(p, CoreDialect::OpenCl));
+ let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" };
+ let hot_args = if p.has_hot() { ", __global const uint* hot" } else { "" };
+ s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask{hot_args}{scratch_args}) {{\n"));
+ let (setup, unit_loop) = persistent_prologue_parts(CoreDialect::OpenCl, p.class.scratch_words_per_lane());
+ if p.has_scratch() {
+ s.push_str(&setup);
+ } else {
+ s.push_str(" uint gid = (uint)get_global_id(0);\n");
+ }
+ s.push_str(" uint lid = (uint)get_local_id(0);\n");
+ if p.has_scratch() {
+ s.push_str("#if IGNEUM_EXCHANGE == 0\n");
+ s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n");
+ s.push_str(" uint xk = 0u;\n");
+ s.push_str("#else\n");
+ s.push_str(" (void)lid;\n");
+ s.push_str("#endif\n");
+ s.push_str(&unit_loop);
+ s.push_str(" uint nonce = baseNonce + gid;\n");
+ s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n");
+ } else {
+ s.push_str(" uint nonce = baseNonce + gid;\n");
+ s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n");
+ s.push_str("#if IGNEUM_EXCHANGE == 0\n");
+ s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n");
+ s.push_str(" uint xk = 0u;\n");
+ s.push_str("#else\n");
+ s.push_str(" (void)lid;\n");
+ s.push_str("#endif\n");
+ }
+ if p.has_wide() {
+ s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n");
+ }
+ for i in 0..8 {
+ s.push_str(&init_line(p, "uint", i));
+ }
+ s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n"));
+ s.push_str(&opencl_instr_lines(p, dataset_log2));
+ s.push_str(&shadow_block(p, CoreDialect::OpenCl));
+ s.push_str(" }\n");
+ s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n");
+ s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n");
+ s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n");
+ if p.has_scratch() {
+ s.push_str(" }\n");
+ }
+ s.push_str("}\n");
+ s.push('\n');
+ s.push_str("#if IGNEUM_EXCHANGE != 0\n");
+ s.push_str("// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the\n");
+ s.push_str("// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a\n");
+ s.push_str("// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.\n");
+ s.push_str("IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {\n");
+ s.push_str(" if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }\n");
+ s.push_str("}\n");
+ s.push_str("#endif\n");
+ s
+}
+
+fn mask_for(dataset_log2: u32) -> u32 {
+ if dataset_log2 >= 32 {
+ u32::MAX
+ } else {
+ (1u32 << dataset_log2) - 1
+ }
+}
+
+const STDINT_BLOCK: &str = "#ifdef __cplusplus\n#include \n#else\n#include \n#endif\n";
+
+/// program.h (`generateProgramHeader`).
+pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String {
+ let key = &ds.key;
+ let dataset_log2 = ds.log2_words;
+ let memhard = ds.memhard().map(|m| &m.params);
+ let mask = mask_for(dataset_log2);
+ let mut s = String::with_capacity(2600);
+ s.push_str(&generated_by(&p.seed_string));
+ s.push_str("// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.\n");
+ s.push_str("// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.\n");
+ s.push_str("#pragma once\n");
+ s.push_str(STDINT_BLOCK);
+ s.push_str("#ifndef IGNEUM_NO_CUDA\n#include \n#endif\n");
+ s.push('\n');
+ s.push_str(&format!("#define IGNEUM_SEED_STRING {}\n", jstr(&p.seed_string)));
+ s.push_str(&format!("#define IGNEUM_SEED_BYTES_HEX {}\n", jstr(&hex_bytes(&p.seed_bytes))));
+ s.push_str(&format!("#define IGNEUM_GENERATOR {}\n", p.generator));
+ s.push_str(&format!("#define IGNEUM_PROGRAM_ATTEMPT {}\n", p.attempt));
+ s.push_str(&format!("#define IGNEUM_PROGRAM_ID {}\n", hex64(p.program_id())));
+ s.push_str(&format!("#define IGNEUM_DAY_STRING {}\n", jstr(day)));
+ s.push_str(&format!("#define IGNEUM_DAY_BYTES_HEX {}\n", jstr(&hex_bytes(&ds.key_bytes))));
+ s.push_str(&format!("#define IGNEUM_DAY0 {}\n", hex(key[0])));
+ s.push_str(&format!("#define IGNEUM_DAY1 {}\n", hex(key[1])));
+ s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n"));
+ s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask)));
+ s.push_str("#define IGNEUM_LANES 32\n");
+ s.push_str(&format!("#define IGNEUM_ITERATIONS {ITERATIONS}\n"));
+ s.push_str(&format!("#define IGNEUM_INSTR_COUNT {INSTR_COUNT}\n"));
+ s.push_str(&format!("#define IGNEUM_LOADS_PER_HASH {}\n", p.loads_per_hash()));
+ s.push_str(&format!("#define IGNEUM_WIDE_LOADS_PER_HASH {}\n", p.wide_loads_per_hash()));
+ s.push_str(&format!("#define IGNEUM_OP_MIX {}\n", jstr(&p.op_mix())));
+ s.push_str(&program_class_header_lines(p));
+ s.push_str(&class_header_lines(p));
+ s.push_str(&state_header_lines(ds));
+ s.push_str(&scratch_header_lines(p));
+ s.push_str(&era_header_lines(p));
+ s.push_str(&hot_header_lines(p));
+ s.push_str(&shadow_header_lines(p));
+ s.push_str("// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)\n");
+ s.push_str(&format!("#define IGNEUM_DATASET_MODE {}\n", if memhard.is_some() { 1 } else { 0 }));
+ s.push('\n');
+ s.push_str(&format!("#define IGNEUM_SEEDW_INIT {{ {} }}\n", join_hex(&p.seed)));
+ if let Some(mp) = memhard {
+ s.push_str(&format!("#define IGNEUM_KEY_INIT {{ {} }}\n", join_hex(&mp.key)));
+ s.push_str(&format!("#define IGNEUM_CACHE_LOG2_WORDS {}\n", mp.shape.cache_log2_words));
+ s.push_str(&format!("#define IGNEUM_CACHE_SEGMENT_LOG2_LINES {CACHE_SEGMENT_LOG2_LINES}\n"));
+ s.push_str(&format!("#define IGNEUM_CACHE_SEGMENTS {}u\n", mp.shape.cache_segments()));
+ s.push_str(&format!("#define IGNEUM_ITEM_ROUNDS {ITEM_ROUNDS}\n"));
+ if mp.shape.mixer_mult != 1 {
+ s.push_str(&format!("#define IGNEUM_MIXER_MULT {} // mixer applications per round and after the last read (class v3, docs/plans/mixer-x4.md)\n", mp.shape.mixer_mult));
+ }
+ if let Some(dp) = &mp.derive {
+ s.push_str(&format!("#define IGNEUM_DERIVE_LEN {} // instructions per round program of the day's item-derivation program (Counter ASIC 3.0 item 2; memhard.h mh_round_0..8)\n", dp.len));
+ s.push_str(&format!("#define IGNEUM_DERIVE_ATTEMPT {}\n", dp.attempt));
+ s.push_str(&format!("#define IGNEUM_DERIVE_FINGERPRINT {}\n", hex64(dp.fingerprint())));
+ s.push_str(&format!("#define IGNEUM_DERIVE_INSTRS_PER_ITEM {}\n", dp.instr_count()));
+ s.push_str(&format!("#define IGNEUM_DERIVE_GPU_OPS_PER_ITEM {}\n", dp.gpu_ops()));
+ s.push_str(&format!("#define IGNEUM_DERIVE_CHIP_OPS_PER_ITEM {}\n", dp.chip_ops()));
+ s.push_str(&format!("#define IGNEUM_DERIVE_MULS_PER_ITEM {}\n", dp.muls()));
+ }
+ s.push_str(&format!(
+ "#define IGNEUM_MIX_ROT_INIT {{ {} }}\n",
+ mp.rot.iter().map(|r| format!("{r}u")).collect::>().join(", ")
+ ));
+ s.push_str(&format!("#define IGNEUM_MIX_MUL_INIT {{ {} }}\n", join_hex(&mp.mul)));
+ s.push_str(&format!("#define IGNEUM_MIX_RC_INIT {{ {} }}\n", join_hex(&mp.rc)));
+ s.push('\n');
+ s.push_str("#ifndef IGNEUM_NO_CUDA\n");
+ s.push_str("// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().\n");
+ s.push_str("cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);\n");
+ if p.class.state {
+ s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems);\n");
+ } else {
+ s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);\n");
+ }
+ if p.has_hot() {
+ s.push_str("cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments);\n");
+ }
+ } else {
+ s.push_str("#ifndef IGNEUM_NO_CUDA\n");
+ s.push_str("// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().\n");
+ s.push_str("cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);\n");
+ }
+ let hot_decl = if p.has_hot() { " const uint32_t* hot," } else { "" };
+ if p.has_scratch() {
+ s.push_str(&format!("cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n"));
+ s.push_str(" uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt);\n");
+ } else {
+ s.push_str(&format!(
+ "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n",
+ ));
+ s.push_str(" uint32_t nonces, uint32_t blockWarps);\n");
+ }
+ s.push_str("cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);\n");
+ s.push_str("#endif\n");
+ s
+}
+
+/// memhard.h (`generateMemhardHeader`): the core in CUDA C++, compiled for host and device.
+pub fn cuda_memhard_header(p: &Program, mp: &MixParams) -> String {
+ let mut s = String::with_capacity(6000);
+ s.push_str(&generated_by(&p.seed_string));
+ s.push_str("// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.\n");
+ s.push_str(
+ "// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).\n",
+ );
+ s.push_str("// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.\n");
+ s.push_str("#pragma once\n");
+ s.push_str(STDINT_BLOCK);
+ s.push_str("#if defined(__CUDACC__)\n");
+ s.push_str("#define IGNEUM_HD __host__ __device__ __forceinline__\n");
+ s.push_str("#elif defined(_MSC_VER) && !defined(__cplusplus)\n");
+ s.push_str("#define IGNEUM_HD static __inline\n");
+ s.push_str("#else\n");
+ s.push_str("#define IGNEUM_HD static inline\n");
+ s.push_str("#endif\n");
+ s.push_str(&emit_memhard_core_layout(mp, CoreDialect::Cuda, p.class.layout()));
+ if p.has_hot() {
+ s.push('\n');
+ s.push_str(&emit_hot_core(p, CoreDialect::Cuda));
+ }
+ s
+}
+
+/// The self-test values a pack carries beside the 96 hashes.
+#[derive(Clone, Debug, Default, PartialEq, Eq)]
+pub struct PackVectors {
+ /// dataset[0..15]
+ pub head: Vec,
+ /// dataset[MASK]
+ pub last: u32,
+ /// 64 sampled dataset indices and their values
+ pub sample_idx: Vec,
+ pub sample_val: Vec,
+ /// cache[0..15] (memory-hard only)
+ pub cache_head: Vec,
+ /// the last cache line (memory-hard only)
+ pub cache_last: Vec,
+ /// FNV-1a 64 over the whole cache (memory-hard only)
+ pub cache_fnv: u64,
+ /// The cache is 2^cache_log2_words words (memory-hard only; 26 under version 2)
+ pub cache_log2_words: u32,
+ /// Hot table (hot packs only): head line, last line, FNV-1a 64 over the whole table
+ pub has_hot: bool,
+ pub hot_head: Vec,
+ pub hot_last: Vec,
+ pub hot_fnv: u64,
+}
+
+/// The base nonces of the three vector warps every pack carries.
+pub const PACK_VECTOR_BASES: [u32; 3] = [0, 4096, 1_000_000];
+
+/// The 64 sampled dataset indices: SplitMix64 seeded with "mhsample", low 32 bits masked.
+pub fn sample_indices(mask: u32) -> Vec {
+ let mut sr = SplitMix64::new(0x6d68_7361_6d70_6c65);
+ (0..64).map(|_| (sr.next() as u32) & mask).collect()
+}
+
+/// vectors.h (`generateVectorsHeader`).
+pub fn vectors_header(
+ p: &Program,
+ bases: &[u32],
+ outs: &[[u64; 32]],
+ v: &PackVectors,
+ mask: u32,
+ source: &str,
+ memhard: bool,
+) -> String {
+ let mut s = String::with_capacity(6000);
+ s.push_str(&generated_by(&p.seed_string));
+ s.push_str(&format!("// Expected outputs: {source}\n"));
+ s.push_str("#pragma once\n");
+ s.push_str(STDINT_BLOCK);
+ s.push('\n');
+ s.push_str(&format!("#define IGNEUM_VEC_WARPS {}\n", bases.len()));
+ s.push_str(&format!(
+ "static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = {{ {} }};\n",
+ bases.iter().map(|b| format!("{b}u")).collect::>().join(", ")
+ ));
+ s.push_str("static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {\n");
+ for (i, o) in outs.iter().enumerate() {
+ s.push_str(&format!(" {{ // base nonce {}\n", bases[i]));
+ for row in 0..4 {
+ s.push_str(" ");
+ s.push_str(&(0..8).map(|c| hex64(o[row * 8 + c])).collect::>().join(", "));
+ s.push_str(if row == 3 { "\n" } else { ",\n" });
+ }
+ s.push_str(if i == outs.len() - 1 { " }\n" } else { " },\n" });
+ }
+ s.push_str("};\n");
+ s.push('\n');
+ s.push_str(&format!("// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] ({mask}).\n"));
+ s.push_str("static const uint32_t IGNEUM_DS_HEAD[16] = {\n");
+ s.push_str(&format!(" {},\n", join_hex(&v.head[..8])));
+ s.push_str(&format!(" {}\n", join_hex(&v.head[8..16])));
+ s.push_str("};\n");
+ s.push_str(&format!("static const uint32_t IGNEUM_DS_LAST_INDEX = {mask}u;\n"));
+ s.push_str(&format!("static const uint32_t IGNEUM_DS_LAST = {};\n", hex(v.last)));
+ s.push_str("// 64 sampled dataset words (index, value) computed on the Mac.\n");
+ s.push_str(&format!("#define IGNEUM_DS_SAMPLES {}\n", v.sample_idx.len()));
+ s.push_str("static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {\n");
+ s.push_str(&format!(" {}\n", v.sample_idx.iter().map(|i| format!("{i}u")).collect::>().join(", ")));
+ s.push_str("};\n");
+ s.push_str("static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {\n");
+ s.push_str(&format!(" {}\n", join_hex(&v.sample_val)));
+ s.push_str("};\n");
+ if memhard {
+ s.push_str(&format!(
+ "// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^{} words.\n",
+ v.cache_log2_words
+ ));
+ s.push_str("static const uint32_t IGNEUM_CACHE_HEAD[16] = {\n");
+ s.push_str(&format!(" {},\n", join_hex(&v.cache_head[..8])));
+ s.push_str(&format!(" {}\n", join_hex(&v.cache_head[8..16])));
+ s.push_str("};\n");
+ s.push_str("static const uint32_t IGNEUM_CACHE_LAST[16] = {\n");
+ s.push_str(&format!(" {},\n", join_hex(&v.cache_last[..8])));
+ s.push_str(&format!(" {}\n", join_hex(&v.cache_last[8..16])));
+ s.push_str("};\n");
+ s.push_str(&format!("static const uint64_t IGNEUM_CACHE_FNV64 = {};\n", hex64(v.cache_fnv)));
+ }
+ if v.has_hot {
+ s.push_str("// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words.\n");
+ s.push_str("static const uint32_t IGNEUM_HOT_HEAD[16] = {\n");
+ s.push_str(&format!(" {},\n", join_hex(&v.hot_head[..8])));
+ s.push_str(&format!(" {}\n", join_hex(&v.hot_head[8..16])));
+ s.push_str("};\n");
+ s.push_str("static const uint32_t IGNEUM_HOT_LAST[16] = {\n");
+ s.push_str(&format!(" {},\n", join_hex(&v.hot_last[..8])));
+ s.push_str(&format!(" {}\n", join_hex(&v.hot_last[8..16])));
+ s.push_str("};\n");
+ s.push_str(&format!("static const uint64_t IGNEUM_HOT_FNV64 = {};\n", hex64(v.hot_fnv)));
+ }
+ s
+}
+
+/// program.json (`generateProgramJSON`). Valid JSON (see the module note about the `"item"` line).
+pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String {
+ let key = &ds.key;
+ let dataset_log2 = ds.log2_words;
+ let memhard = ds.memhard().map(|m| &m.params);
+ let mask = mask_for(dataset_log2);
+ let mut s = String::with_capacity(14000);
+ s.push_str("{\n");
+ s.push_str(" \"format\": \"igneum-program-pack-3\",\n");
+ s.push_str(&format!(" \"generator\": {},\n", p.generator));
+ s.push_str(&format!(" \"attempt\": {},\n", p.attempt));
+ s.push_str(&format!(" \"program_id\": {},\n", jhex64(p.program_id())));
+ s.push_str(" \"program_id_derivation\": \"FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32\",\n");
+ s.push_str(&format!(
+ " \"dataset_mode\": {},\n",
+ jstr(if memhard.is_some() { "memory-hard" } else { "closed-form" })
+ ));
+ s.push_str(&format!(" \"seed\": {},\n", jstr(&p.seed_string)));
+ s.push_str(&format!(" \"seed_bytes\": {},\n", jstr(&hex_bytes(&p.seed_bytes))));
+ s.push_str(&format!(" \"seed_words\": [{}],\n", join_jhex(&p.seed)));
+ s.push_str(" \"seed_derivation\": \"seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32\",\n");
+ s.push_str(&format!(" \"generator_rule\": \"version {GENERATOR_VERSION}: exactly {LOAD_SLOTS} load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried\",\n"));
+ s.push_str(" \"lanes\": 32,\n");
+ s.push_str(" \"registers\": 8,\n");
+ s.push_str(&format!(" \"iterations\": {ITERATIONS},\n"));
+ s.push_str(&format!(" \"instruction_count\": {INSTR_COUNT},\n"));
+ s.push_str(&format!(" \"loads_per_hash\": {},\n", p.loads_per_hash()));
+ if p.program_class() != ProgramClass::V2 {
+ s.push_str(&format!(" \"program_class\": {},\n", jstr(p.program_class().name())));
+ if p.program_class() == ProgramClass::V4 {
+ s.push_str(&format!(" \"sub_version\": {},\n", PROGRAM_SUBVERSION_V4));
+ }
+ if let Some(era) = &p.era_bytes {
+ s.push_str(&format!(" \"era_seed_bytes\": {},\n", jstr(&hex_bytes(era))));
+ }
+ }
+ if let Some(l) = ds.leaves() {
+ s.push_str(" \"state\": {\n");
+ s.push_str(&format!(" \"block\": {},\n", jstr(&hex_bytes(&l.block))));
+ s.push_str(&format!(" \"block_number\": {},\n", l.number));
+ s.push_str(&format!(" \"root\": {},\n", jstr(&hex_bytes(&l.root))));
+ s.push_str(&format!(" \"leaves\": {},\n", l.n()));
+ s.push_str(&format!(" \"records\": {},\n", l.records_total));
+ s.push_str(&format!(" \"sampled\": {},\n", l.sampled));
+ s.push_str(&format!(" \"leaves_fnv1a64\": {},\n", jhex64(l.fnv1a64())));
+ s.push_str(" \"leaf_derivation\": \"leaves[i] = Blake2b-512('igneum-sd1/' || root || i_le32 || record_i) as 16 little-endian words; item t XORs leaves[t mod leaves] into its 16 initial words before the first mixer\",\n");
+ s.push_str(" \"file\": \"leaves.bin\"\n");
+ s.push_str(" },\n");
+ }
+ if !p.class.is_v2() {
+ let c = p.width_counts();
+ s.push_str(&format!(" \"load_class\": {},\n", jstr(&p.class.name())));
+ if p.class.derive_len != 0 {
+ s.push_str(&format!(" \"derive_len\": {},\n", p.class.derive_len));
+ s.push_str(" \"derive\": \"Counter ASIC 3.0 item 2 (6 October 2026, docs/plans/counter-asic-3-derivation.md; a prototype, not class v3): the nine mixer slots of the item derivation each run a straight-line program of derive_len instructions drawn from the day key stream after the 40 mixer draws, four draws per instruction (op roll below(100), destination roll below(15), third-register roll below(14), the immediate next()); every instruction reads the register the previous one wrote (s[0] first) and writes another; twelve forms, each a bijection on the state; the 8 dependent cache reads per item unchanged; the acceptance test of derive.rs (every register written per round program, 8 distinct rotations, the x8 mixer's operation and multiply counts as floors) rejects a draw and the next attempt continues the stream\",\n");
+ }
+ if p.class.mixer_mult != 1 || p.class.growth {
+ s.push_str(&format!(" \"mixer_mult\": {},\n", p.class.mixer_mult));
+ s.push_str(&format!(" \"cache_growth\": {},\n", p.class.growth));
+ s.push_str(&format!(" \"mixer\": \"class v3 (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): every mixer application of the item derivation is {} applications with round keys (r * {} + j + 1) * 0x9E3779B9, the 8 dependent cache reads per item unchanged; cache growth rule option C: cache words = 2^(26 + doublings(day)), dataset words = 2^(genesis_log2 + doublings(day)), doublings(day) = floor(log2(1 + day / 1460)) for day = days since genesis\",\n", p.class.mixer_mult, p.class.mixer_mult));
+ }
+ s.push_str(&format!(" \"load_slots\": {},\n", p.class.load_slots));
+ s.push_str(&format!(" \"load_mix_percent_4_16_64\": [{}, {}, {}],\n", p.class.mix[0], p.class.mix[1], p.class.mix[2]));
+ s.push_str(&format!(" \"load_width_counts_4_16_64\": [{}, {}, {}],\n", c[0], c[1], c[2]));
+ s.push_str(&format!(" \"bytes_per_hash\": {},\n", p.bytes_per_hash()));
+ if p.has_scratch() {
+ s.push_str(&format!(" \"scratch_ops_per_hash\": {},\n", p.scratch_ops_per_hash()));
+ s.push_str(&format!(" \"scratch_kib_per_warp\": {},\n", p.class.scratch_kb));
+ s.push_str(&format!(" \"scratch\": \"variant 5 (measurement only): persistent warps; a {kb} KiB scratch per warp of {slots} 16-byte slots per lane (lane-major); slot = src & 0x{smask:x}; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)\",\n", kb = p.class.scratch_kb, slots = p.class.scratch_slots_per_lane(), smask = p.class.scratch_slot_mask()));
+ }
+ s.push_str(&format!(" \"wide_load\": \"read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, {FOLD_ROT}) * 0x{FOLD_MUL:08x}) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots\",\n"));
+ if let Some(e) = p.class.era {
+ s.push_str(" \"era\": {\n");
+ s.push_str(&format!(" \"label\": {},\n", jstr(&e.label())));
+ s.push_str(&format!(" \"seed_words\": [{}],\n", join_jhex(&e.words)));
+ s.push_str(" \"draw\": \"docs/plans/era-layout.md 1.1: SplitMix64 seeded with seed_words[0] | seed_words[1] << 32 of seed_words_from_bytes('igneum-era/' || n_le64 || E_n); width = allowed[below(|allowed|)], stride_mul = low32(next()) | 1, stride_rot = 1 + below(31), then four next() draws for a partial Fisher-Yates over positions log2(W)..15 of which 4 - log2(W) are used\",\n");
+ s.push_str(&format!(" \"allowed_widths\": [{}],\n", e.allowed_set().iter().map(|w| w.to_string()).collect::>().join(", ")));
+ s.push_str(&format!(" \"width_words\": {},\n", e.width_words));
+ s.push_str(&format!(" \"stride_mul\": {},\n", jhex(e.stride_mul)));
+ s.push_str(&format!(" \"stride_rot\": {},\n", e.stride_rot));
+ s.push_str(&format!(" \"interleave\": [{}, {}, {}, {}],\n", e.pos[0], e.pos[1], e.pos[2], e.pos[3]));
+ s.push_str(" \"address\": \"y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words\",\n");
+ s.push_str(" \"windows\": \"per instruction, after the width roll: win = below(3), off = low32(next()) & (2^win - 1); used on a load slot (the instruction's win and off fields)\",\n");
+ s.push_str(" \"dataset_word\": \"dataset[w] = item(t(w))[j(w)]: j(w) gathers the bits of w at the interleave positions, t(w) is w with those bits removed\",\n");
+ s.push_str(" \"program_id_suffix\": \"'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]\"\n");
+ s.push_str(" },\n");
+ }
+ if let Some(h) = p.class.hot {
+ let hk = hot_key(&p.seed_bytes);
+ s.push_str(&format!(
+ " \"hot_table\": {{\"mb\": {}, \"words\": {}, \"segments\": {}, \"slots\": {}, \"form\": {}, \"dataset_slots\": {}, \"hot_loads_per_hash\": {}, \"key\": [{}], \"key_derivation\": \"seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch\", \"tag\": [{}], \"chain\": \"the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0\", \"load\": \"dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)\", \"slots_rule\": \"the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected\", \"acceptance_stand_in\": \"dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])\", \"program_id\": \"the read-width id with 'hot/' || mb || k appended\", \"spec\": \"docs/plans/hot-table.md\"}},\n",
+ h.mb,
+ hot_words(h.mb as u32),
+ hot_segments(h.mb as u32),
+ h.k,
+ jstr(if h.added { "added: k load slots added beside the class's, the dataset loads unchanged" } else { "replaced: k of the class's load slots read the table" }),
+ p.class.dataset_slots(),
+ p.hot_loads_per_hash(),
+ join_jhex(&hk),
+ join_jhex(&HOT_TAG)
+ ));
+ }
+ }
+ s.push_str(&format!(
+ " \"op_mix\": {{{}}},\n",
+ p.histogram().iter().map(|(n, c)| format!("{}: {c}", jstr(n))).collect::>().join(", ")
+ ));
+ s.push_str(" \"register_init\": \"for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]\",\n");
+ s.push_str(" \"splitmix32\": \"x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16\",\n");
+ s.push_str(
+ " \"iteration\": \"sel = r0 sampled once at the top of each iteration, then all instructions in order\",\n",
+ );
+ s.push_str(" \"output\": \"lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo\",\n");
+ s.push_str(" \"op_semantics\": {\n");
+ s.push_str(" \"add\": \"dst = dst + src + (bit `bit` of sel ? imm2 : imm)\",\n");
+ s.push_str(" \"sub\": \"dst = dst - src\",\n");
+ s.push_str(" \"mul\": \"dst = dst * src (low 32)\",\n");
+ s.push_str(" \"mulhi\": \"dst = high 32 bits of dst * src\",\n");
+ s.push_str(" \"xor\": \"dst = dst ^ src\",\n");
+ s.push_str(" \"or\": \"dst = dst | src\",\n");
+ s.push_str(" \"rotl\": \"dst = rotl(dst, rot), rot in 1..31\",\n");
+ s.push_str(" \"rotr\": \"dst = rotr(dst, src & 31)\",\n");
+ s.push_str(" \"mad\": \"dst = src * src2 + dst\",\n");
+ s.push_str(
+ " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n",
+ );
+ s.push_str(" \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n");
+ s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\"");
+ if p.has_hot() {
+ s.push_str(",\n \"hot\": \"dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)\"");
+ }
+ s.push('\n');
+ s.push_str(" },\n");
+ s.push_str(" \"dataset\": {\n");
+ s.push_str(&format!(" \"log2_words\": {dataset_log2},\n"));
+ s.push_str(&format!(" \"bytes\": {},\n", 1u64 << (dataset_log2 as u64 + 2)));
+ s.push_str(&format!(" \"mask\": {},\n", jhex(mask)));
+ s.push_str(&format!(" \"day\": {},\n", jstr(day)));
+ s.push_str(&format!(" \"day_bytes\": {},\n", jstr(&hex_bytes(&ds.key_bytes))));
+ s.push_str(" \"day_words_from\": \"seed_words_from_bytes(day_bytes)\",\n");
+ s.push_str(&format!(" \"d0\": {},\n", jhex(key[0])));
+ s.push_str(&format!(" \"d1\": {},\n", jhex(key[1])));
+ if let Some(mp) = memhard {
+ s.push_str(" \"mode\": \"memory-hard\",\n");
+ s.push_str(" \"spec\": \"proto-metal/MEMHARD.md\",\n");
+ s.push_str(&format!(" \"key\": [{}],\n", join_jhex(&mp.key)));
+ s.push_str(" \"key_derivation\": \"the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]\",\n");
+ let shape = &mp.shape;
+ s.push_str(&format!(
+ " \"cache\": {{\"log2_words\": {}, \"bytes\": {}, \"line_words\": 16, \"segment_lines\": {CACHE_LINES_PER_SEGMENT}, \"segments\": {}, \"block\": \"ChaCha{CHACHA_ROUNDS} core + feed-forward, rotations 16 12 8 7\", \"sigma\": [{}], \"tag\": [{}], \"chain\": \"in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0\"}},\n",
+ shape.cache_log2_words,
+ shape.cache_words() as u64 * 4,
+ shape.cache_segments(),
+ join_jhex(&CHACHA_SIGMA),
+ join_jhex(&CACHE_TAG)
+ ));
+ s.push_str(&format!(
+ " \"mixer\": {{\"draw\": \"SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())\", \"rot\": [{}], \"mul\": [{}], \"rc\": [{}], \"round\": \"for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]\", \"quarter_round\": \"a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)\"}},\n",
+ mp.rot.iter().map(|r| r.to_string()).collect::>().join(", "),
+ join_jhex(&mp.mul),
+ join_jhex(&mp.rc)
+ ));
+ // The Swift writes jhex(cacheLineMask) here, which breaks the JSON. We write the bare literal.
+ if let Some(dp) = &mp.derive {
+ s.push_str(&format!(
+ " \"derive_len\": {},\n \"derive_attempt\": {},\n \"derive_fingerprint\": {},\n \"derive_op_mix\": {},\n \"derive_instrs_per_item\": {},\n \"derive_gpu_ops_per_item\": {},\n \"derive_chip_ops_per_item\": {},\n \"derive_muls_per_item\": {},\n",
+ dp.len, dp.attempt, jhex64(dp.fingerprint()), jstr(&dp.op_mix()), dp.instr_count(), dp.gpu_ops(), dp.chip_ops(), dp.muls()
+ ));
+ s.push_str(&format!(
+ " \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..{}: s = P_r(s); line = s[0] & 0x{:08x}; s[i] ^= cache[line * 16 + i]; then s = P_{ITEM_ROUNDS}(s); item(t) = s; P_r is round program r below (c = the register the previous instruction wrote, s[0] first): add d += c; sub d -= c; xor d ^= c; mul d *= (c | 1); rot d = rotl(d, k) + c; xrot d = rotl(d ^ c, k); addc d += c + imm; xorc d ^= c ^ imm; mulc d = (d ^ c) * imm; mulc2 d = d * imm + c; andx d ^= (c & b); orx d += (c | b)\",\n",
+ ITEM_ROUNDS - 1,
+ shape.cache_line_mask()
+ ));
+ s.push_str(" \"programs\": [\n");
+ for (r, prog) in dp.rounds.iter().enumerate() {
+ s.push_str(" [");
+ s.push_str(&prog.iter().map(|i| jstr(&derive_instr_line(i))).collect::>().join(", "));
+ s.push_str(if r + 1 < dp.rounds.len() { "],\n" } else { "]\n" });
+ }
+ s.push_str(" ],\n");
+ } else if shape.mixer_mult == 1 {
+ s.push_str(&format!(
+ " \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..{}: s = M_r(s); line = s[0] & 0x{:08x}; s[i] ^= cache[line * 16 + i]; then s = M_{ITEM_ROUNDS}(s); item(t) = s\",\n",
+ ITEM_ROUNDS - 1,
+ shape.cache_line_mask()
+ ));
+ } else {
+ let m = shape.mixer_mult;
+ s.push_str(&format!(
+ " \"mixer_mult\": {m},\n \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..{}: for j in 0..{}: s = M(s, rk = (r * {m} + j + 1) * 0x9E3779B9); line = s[0] & 0x{:08x}; s[i] ^= cache[line * 16 + i]; then for j in 0..{}: s = M(s, rk = ({} + j + 1) * 0x9E3779B9); item(t) = s\",\n",
+ ITEM_ROUNDS - 1,
+ m - 1,
+ shape.cache_line_mask(),
+ m - 1,
+ ITEM_ROUNDS as u32 * m
+ ));
+ }
+ s.push_str(" \"word\": \"dataset[w] = item(w >> 4)[w & 15]\"\n");
+ } else {
+ s.push_str(" \"mode\": \"closed-form\",\n");
+ s.push_str(" \"formula\": \"x = i ^ d0; x *= 0x9E3779B1; x ^= x>>15; x += d1; x *= 0x85EBCA77; x ^= x>>13; x *= 0xC2B2AE3D; x ^= x>>16 (all mod 2^32)\"\n");
+ }
+ s.push_str(" },\n");
+ if let Some(sh) = p.class.shadow {
+ s.push_str(&format!(
+ " \"shadow\": {{\"instrs\": {}, \"reps\": {}, \"instrs_per_hash\": {}, \"op_mix\": {{{}}}, \"rule\": \"Counter ASIC 3.0 item 8 (docs/analysis/latency-shadow-2026-10-06.md): after the 64 base instructions the program stream draws instrs more ALU instructions (the non-load table, nine draws each, the source as on an ALU slot); the block runs reps times at the end of every iteration with the iteration's sel; the acceptance rule interprets the base program only\", \"program_id_suffix\": \"'shadow/' || instrs_le16 || reps_le16\", \"instructions\": [\n",
+ sh.instrs,
+ sh.reps,
+ p.shadow_instrs_per_hash(),
+ p.shadow_histogram().iter().map(|(n, c)| format!("{}: {c}", jstr(n))).collect::>().join(", ")
+ ));
+ let n = p.shadow.len();
+ for (k, ins) in p.shadow.iter().enumerate() {
+ s.push_str(&format!(
+ " {{\"i\": {k}, \"op\": {}, \"dst\": {}, \"src\": {}, \"src2\": {}, \"imm\": {}, \"imm2\": {}, \"rot\": {}, \"bit\": {}, \"mask\": {}}}{}\n",
+ jstr(ins.op.name()),
+ ins.dst,
+ ins.src,
+ ins.src2,
+ jhex(ins.imm),
+ jhex(ins.imm2),
+ ins.rot,
+ ins.bit,
+ ins.mask,
+ if k + 1 < n { "," } else { "" }
+ ));
+ }
+ s.push_str(" ]},\n");
+ }
+ s.push_str(" \"instructions\": [\n");
+ let n = p.instrs.len();
+ for (k, ins) in p.instrs.iter().enumerate() {
+ if !p.class.is_v2() {
+ let era_fields = if p.class.era.is_some() { format!(", \"win\": {}, \"off\": {}", ins.win, ins.off) } else { String::new() };
+ s.push_str(&format!(
+ " {{\"i\": {k}, \"op\": {}, \"dst\": {}, \"src\": {}, \"src2\": {}, \"imm\": {}, \"imm2\": {}, \"rot\": {}, \"bit\": {}, \"mask\": {}, \"width\": {}{era_fields}}}",
+ jstr(ins.op.name()),
+ ins.dst,
+ ins.src,
+ ins.src2,
+ jhex(ins.imm),
+ jhex(ins.imm2),
+ ins.rot,
+ ins.bit,
+ ins.mask,
+ ins.width
+ ));
+ s.push_str(if k + 1 < n { ",\n" } else { "\n" });
+ continue;
+ }
+ s.push_str(&format!(
+ " {{\"i\": {k}, \"op\": {}, \"dst\": {}, \"src\": {}, \"src2\": {}, \"imm\": {}, \"imm2\": {}, \"rot\": {}, \"bit\": {}, \"mask\": {}}}",
+ jstr(ins.op.name()),
+ ins.dst,
+ ins.src,
+ ins.src2,
+ jhex(ins.imm),
+ jhex(ins.imm2),
+ ins.rot,
+ ins.bit,
+ ins.mask
+ ));
+ s.push_str(if k == n - 1 { "\n" } else { ",\n" });
+ }
+ s.push_str(" ]\n}\n");
+ s
+}
+
+/// vectors.json (`generateVectorsJSON`).
+pub fn vectors_json(
+ p: &Program,
+ day: &str,
+ dataset_log2: u32,
+ bases: &[u32],
+ outs: &[[u64; 32]],
+ v: &PackVectors,
+ mask: u32,
+ source: &str,
+ memhard: bool,
+) -> String {
+ let mut s = String::with_capacity(6500);
+ s.push_str("{\n");
+ s.push_str(&format!(" \"seed\": {},\n", jstr(&p.seed_string)));
+ s.push_str(&format!(" \"day\": {},\n", jstr(day)));
+ s.push_str(&format!(" \"dataset_mode\": {},\n", jstr(if memhard { "memory-hard" } else { "closed-form" })));
+ s.push_str(&format!(" \"dataset_log2_words\": {dataset_log2},\n"));
+ s.push_str(&format!(" \"mask\": {},\n", jhex(mask)));
+ s.push_str(" \"lanes\": 32,\n");
+ s.push_str(&format!(" \"source\": {},\n", jstr(source)));
+ s.push_str(" \"warps\": [\n");
+ for (i, o) in outs.iter().enumerate() {
+ s.push_str(&format!(" {{\"base_nonce\": {}, \"expected\": [\n", bases[i]));
+ for row in 0..4 {
+ s.push_str(" ");
+ s.push_str(&(0..8).map(|c| jhex64(o[row * 8 + c])).collect::>().join(", "));
+ s.push_str(if row == 3 { "\n" } else { ",\n" });
+ }
+ s.push_str(if i == outs.len() - 1 { " ]}\n" } else { " ]},\n" });
+ }
+ s.push_str(" ],\n");
+ s.push_str(&format!(" \"dataset_head\": [{}],\n", join_jhex(&v.head)));
+ s.push_str(&format!(" \"dataset_last_index\": {mask},\n"));
+ s.push_str(&format!(" \"dataset_last\": {},\n", jhex(v.last)));
+ s.push_str(&format!(
+ " \"dataset_samples\": [{}]",
+ v.sample_idx
+ .iter()
+ .zip(v.sample_val.iter())
+ .map(|(i, val)| format!("{{\"index\": {i}, \"value\": {}}}", jhex(*val)))
+ .collect::>()
+ .join(", ")
+ ));
+ if memhard {
+ s.push_str(&format!(",\n \"cache_head\": [{}],\n", join_jhex(&v.cache_head)));
+ s.push_str(&format!(" \"cache_last_line\": [{}],\n", join_jhex(&v.cache_last)));
+ s.push_str(&format!(" \"cache_fnv1a64\": {}", jhex64(v.cache_fnv)));
+ if v.has_hot {
+ s.push_str(&format!(",\n \"hot_head\": [{}],\n", join_jhex(&v.hot_head)));
+ s.push_str(&format!(" \"hot_last_line\": [{}],\n", join_jhex(&v.hot_last)));
+ s.push_str(&format!(" \"hot_fnv1a64\": {}", jhex64(v.hot_fnv)));
+ }
+ s.push('\n');
+ } else {
+ s.push('\n');
+ }
+ s.push_str("}\n");
+ s
+}
+
+/// A program pack: the files `--export-pack` writes, as (name, text).
+pub struct Pack {
+ pub files: Vec<(String, String)>,
+ /// Binary files beside the texts: `leaves.bin` of a class v5 pack (empty for every other pack).
+ pub binaries: Vec<(String, Vec)>,
+ pub bases: Vec,
+ pub outs: Vec<[u64; 32]>,
+ pub vectors: PackVectors,
+}
+
+impl Pack {
+ pub fn write_to(&self, dir: &std::path::Path) -> std::io::Result<()> {
+ std::fs::create_dir_all(dir)?;
+ for (name, text) in &self.files {
+ std::fs::write(dir.join(name), text)?;
+ }
+ for (name, bytes) in &self.binaries {
+ std::fs::write(dir.join(name), bytes)?;
+ }
+ Ok(())
+ }
+}
+
+/// Build the whole pack for an epoch: the three vector warps from the CPU interpreter, the self-test words,
+/// and every source file. The `source` string says where the vectors came from.
+pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack {
+ let p = &epoch.program;
+ let ds: &DatasetSource = &epoch.dataset;
+ let mask = ds.mask;
+ let memhard = ds.memhard().map(|m| &m.params);
+ let bases = PACK_VECTOR_BASES.to_vec();
+ let outs: Vec<[u64; 32]> = bases.iter().map(|&b| epoch.hash_warp(b)).collect();
+ // the self-test words under the program's layout (era layout; linear for every other class)
+ let mut v = PackVectors {
+ head: (0..16).map(|i| epoch.dataset_word(i)).collect(),
+ last: epoch.dataset_word(mask),
+ sample_idx: sample_indices(mask),
+ ..Default::default()
+ };
+ v.sample_val = v.sample_idx.iter().map(|&i| epoch.dataset_word(i)).collect();
+ if let Some(m) = ds.memhard() {
+ let w = m.cache.words();
+ v.cache_head = w[..16].to_vec();
+ v.cache_last = w[w.len() - 16..].to_vec();
+ v.cache_fnv = m.cache.fnv1a64();
+ v.cache_log2_words = m.shape().cache_log2_words;
+ }
+ if let Some(h) = &ds.hot {
+ let w = h.words();
+ v.has_hot = true;
+ v.hot_head = w[..16].to_vec();
+ v.hot_last = w[w.len() - 16..].to_vec();
+ v.hot_fnv = h.fnv1a64();
+ }
+ let is_mh = memhard.is_some();
+ let mut files = vec![
+ ("program.json".to_string(), program_json(p, day, ds)),
+ ("vectors.json".to_string(), vectors_json(p, day, ds.log2_words, &bases, &outs, &v, mask, source, is_mh)),
+ ("kernel.cu".to_string(), cuda_kernel_at(p, memhard, ds.log2_words)),
+ ("kernel.cl".to_string(), opencl_kernel_at(p, memhard, ds.log2_words)),
+ ("program.h".to_string(), program_header(p, day, ds)),
+ ("vectors.h".to_string(), vectors_header(p, &bases, &outs, &v, mask, source, is_mh)),
+ ("program.metal".to_string(), metal_program(p, ds.log2_words, LoadSource::Stored)),
+ // Header-bound kernels (3 October 2026, bind.rs): new files, the seven above are unchanged.
+ ("program_bound.metal".to_string(), metal_program_bound(p, ds.log2_words)),
+ ("kernel_bound.cu".to_string(), cuda_kernel_bound_at(p, memhard, ds.log2_words)),
+ ("kernel_bound.cl".to_string(), opencl_kernel_bound_at(p, memhard, ds.log2_words)),
+ ];
+ if let Some(mp) = memhard {
+ files.push(("memhard.h".to_string(), cuda_memhard_header(p, mp)));
+ files.push(("memhard.metal".to_string(), metal_memhard_for(p, mp)));
+ }
+ let binaries = match ds.leaves() {
+ Some(l) => vec![("leaves.bin".to_string(), l.bytes())],
+ None => Vec::new(),
+ };
+ Pack { files, binaries, bases, outs, vectors: v }
+}
+
+/// The dataset mode a pack was written in, from its program.json text (no JSON parser needed).
+pub fn pack_mode_from_json(program_json: &str) -> DatasetMode {
+ if program_json.contains("\"dataset_mode\": \"memory-hard\"") {
+ DatasetMode::MemoryHard
+ } else {
+ DatasetMode::ClosedForm
+ }
+}
diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/generator.rs b/tools/attack/adv-accept-v5/igneum-pow/src/generator.rs
new file mode 100644
index 000000000..46955d6c0
--- /dev/null
+++ b/tools/attack/adv-accept-v5/igneum-pow/src/generator.rs
@@ -0,0 +1,2699 @@
+//! The program generator: 64 integer instructions over 8 x u32 lane registers, run for 8 iterations.
+//!
+//! Generator version 2 (adopted 4 October 2026 from `docs/analysis/weak-program-census-2026-10-03.md`,
+//! spec 01 sections 1.4.2, 1.4.3 and 1.4.6):
+//!
+//! * G1, exact load count: every program has exactly [`LOAD_SLOTS`] (16) `load` instructions, drawn first as a
+//! uniform 16-subset of instruction slots 1..63 by a partial Fisher-Yates over the program stream. The other
+//! 48 ops come from the ten non-load families with the weights of [`NONLOAD_WEIGHTS`] (sum 75).
+//! * G2, fresh source: on a load slot the source register is drawn from `E`, the registers other than `dst` that
+//! an earlier instruction of this program has written and that no later load has read. A load's address is then
+//! a value produced in this iteration that no earlier load used, so no load of a hash repeats an earlier load's
+//! address, across the iteration boundary included. If `E` is empty the source is drawn as on an ALU slot and
+//! the acceptance rule of [`crate::accept`] rejects the program.
+//! * R, acceptance: a candidate must pass [`crate::accept::check`]. A rejected candidate is replaced by the next
+//! attempt, `seed_words_from_bytes(program_seed || k_le32)` for `k = 1, 2, ...` (attempt 0 is the bare seed),
+//! so every node derives the same program from the same seed.
+//!
+//! The retired version 1 generator (op rolled per instruction with a 25 percent load weight, no acceptance) is
+//! kept as [`generate_v1`] for the census tool and the lever measurements of `proto-metal/MEMHARD.md`. Its
+//! programs are not the lottery hash and no pack or vector of version 1 is current.
+//!
+//! Era layout (5 October 2026, Counter ASIC 2.0 layers 4 and 8, `docs/plans/era-layout.md`; NOT the lottery hash,
+//! behind [`LoadClass::era`]): [`EraParams`] drawn from the era seed `E_n` by [`era_draw`] (the load width, a stride
+//! multiplier and rotation, the interleave of item words over the dataset), and per load site two more draws (a
+//! window of the dataset: a half, a quarter or all of it, at a drawn offset). The load address is
+//! [`crate::verify::load_index`]. An era class takes 12 draws per instruction, so its stream differs from version 2.
+//!
+//! Read-width experiment (5 October 2026, gate 1, `docs/plans/read-width.md`; NOT the lottery hash, behind
+//! [`LoadClass`]): a program class whose `load` reads `W` bytes (4, 16 or 64: 1, 4 or 16 words, aligned to `W`)
+//! and folds every word into `dst` (`verify::fold_words`), with the width fixed per class or drawn per load from
+//! an era-fixed mix. The default class [`LoadClass::V2`] is the generator above, draw for draw and byte for byte;
+//! every other class takes one extra draw per instruction (the width roll), so its program stream differs from
+//! version 2 and its program id carries the class.
+//!
+//! Hot-table experiment (5 October 2026, Counter ASIC 2.0 layer 5, `docs/plans/hot-table.md`; NOT the lottery hash,
+//! behind [`LoadClass::hot`]): `k` of the load slots read a second table `H` of `S` MiB derived from the epoch seed
+//! ([`crate::memhard::HotTable`]) at `H[mulhi(src, words)]` with the plain one-word fold. The hot slots are the
+//! first `k` drawn load slots after the scratch slots (a uniform `k`-subset, no extra draw), so a class with the
+//! version 2 widths and no scratch takes the version 2 stream exactly ([`LoadClass::takes_width_roll`]).
+
+use crate::accept::{check, Reject};
+use crate::seed::{fnv1a64, program_rng, seed_words_from_bytes, SplitMix64};
+
+/// Iterations of the instruction list per hash.
+pub const ITERATIONS: usize = 8;
+/// Instructions per program.
+pub const INSTR_COUNT: usize = 64;
+/// Lanes per verification unit (one SIMD group / warp).
+pub const LANES: usize = 32;
+/// The generator version written into every pack and program id. Version 1 programs never mix with these.
+pub const GENERATOR_VERSION: u32 = 2;
+/// Load instructions per program under version 2 (G1): 128 loads per hash, 4,096 items per 32-lane unit.
+pub const LOAD_SLOTS: usize = 16;
+/// Attempts before an implementation may treat the seed as a consensus fault (spec 01 section 1.4.6). At the
+/// measured 5.14 percent rejection rate the chance of 32 consecutive rejections is below 2^-136.
+pub const MAX_ATTEMPTS: u32 = 32;
+
+/// The attempt cap of class v4 sub-version 2 (AP-F8-2, 7 October 2026): rules (a') and (c') reject about two thirds of
+/// candidates, so 32 attempts exhaust with probability about (2/3)^32, 2e-6 per epoch seed, one epoch no node could
+/// draw every few decades at one epoch an hour (seen at chain-shaped seed igneum-f9/331672). At 256 attempts the
+/// exhaustion probability is (2/3)^256, under 1e-45; the cost of a rejected attempt is one draw and the 64-unit check,
+/// about 2 ms on one core, so the worst case is half a second. Keyed on the class v4 shape, so v2 and v3 keep 32.
+pub const MAX_ATTEMPTS_V4: u32 = 256;
+
+/// The attempt cap of a class: [`MAX_ATTEMPTS_V4`] for the class v4 shape, [`MAX_ATTEMPTS`] otherwise.
+pub fn max_attempts_for(class: &LoadClass) -> u32 {
+ if crate::accept::is_class_v4_shape(class) { MAX_ATTEMPTS_V4 } else { MAX_ATTEMPTS }
+}
+/// Domain tag of the program id.
+pub const PROGRAM_ID_TAG: &[u8] = b"igneum-program/";
+
+#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
+pub enum Op {
+ Add,
+ Sub,
+ Mul,
+ MulHi,
+ Xor,
+ Or,
+ Rotl,
+ Rotr,
+ Mad,
+ Shfl,
+ Load,
+ /// Warp-coalesced load (lever b of the version 1 generator). Never emitted by version 2.
+ WLoad,
+ /// Scratch read-modify-write (read-width experiment, variant 5, 5 October 2026): a 16-byte slot of the lane's
+ /// own 32 KiB of the warp's 1 MiB scratch, read, folded into dst, rewritten. Never emitted by version 2.
+ Scratch,
+ /// Hot-table load (hot-table experiment, 5 October 2026): `dst = dst XOR H[mulhi(src, HOT_WORDS)]`, one word
+ /// of the epoch's `S` MiB table. Never emitted by version 2.
+ Hot,
+}
+
+impl Op {
+ /// The name used in program.json, kernel comments and the op mix.
+ pub fn name(self) -> &'static str {
+ match self {
+ Op::Add => "add",
+ Op::Sub => "sub",
+ Op::Mul => "mul",
+ Op::MulHi => "mulhi",
+ Op::Xor => "xor",
+ Op::Or => "or",
+ Op::Rotl => "rotl",
+ Op::Rotr => "rotr",
+ Op::Mad => "mad",
+ Op::Shfl => "shfl",
+ Op::Load => "load",
+ Op::WLoad => "wload",
+ Op::Scratch => "scratch",
+ Op::Hot => "hot",
+ }
+ }
+
+ pub fn from_name(s: &str) -> Option {
+ Some(match s {
+ "add" => Op::Add,
+ "sub" => Op::Sub,
+ "mul" => Op::Mul,
+ "mulhi" => Op::MulHi,
+ "xor" => Op::Xor,
+ "or" => Op::Or,
+ "rotl" => Op::Rotl,
+ "rotr" => Op::Rotr,
+ "mad" => Op::Mad,
+ "shfl" => Op::Shfl,
+ "load" => Op::Load,
+ "wload" => Op::WLoad,
+ "scratch" => Op::Scratch,
+ "hot" => Op::Hot,
+ _ => return None,
+ })
+ }
+
+ /// An injecting op: bijective in `dst` and bringing another register (or the dataset) in. The acceptance
+ /// rule's part (b) requires one such write per register.
+ pub fn injects(self) -> bool {
+ matches!(self, Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl | Op::Load | Op::WLoad | Op::Scratch | Op::Hot)
+ }
+
+ /// A memory operation: the fresh-source rule, the acceptance tests and the load count treat the scratch
+ /// read-modify-write and the hot-table load as loads (each is one of the program's 128 memory operations).
+ pub fn is_load(self) -> bool {
+ matches!(self, Op::Load | Op::WLoad | Op::Scratch | Op::Hot)
+ }
+}
+
+/// One instruction. Every field is drawn for every instruction whether the op uses it or not, so the
+/// draw stream is identical for every op.
+#[derive(Clone, Copy, Debug, PartialEq, Eq)]
+pub struct Instr {
+ pub op: Op,
+ /// Destination register 0..7.
+ pub dst: u8,
+ /// Source register 0..7, never equal to `dst`.
+ pub src: u8,
+ /// Second source (mad only).
+ pub src2: u8,
+ /// Add immediate A.
+ pub imm: u32,
+ /// Add immediate B.
+ pub imm2: u32,
+ /// rotl amount 1..31.
+ pub rot: u32,
+ /// Selector bit of r0 for add, 0..31.
+ pub bit: u8,
+ /// Shuffle xor mask: 1, 2, 4, 8 or 16.
+ pub mask: u8,
+ /// Words read by a `load`: 1 (the lottery hash, 4 bytes), 4 or 16 (the read-width experiment). 1 on every
+ /// other op.
+ pub width: u8,
+ /// Era layout, layer 8: the window shrink of this load site, 0..2 (the dataset, a half, a quarter). 0 on every
+ /// op of every other class.
+ pub win: u8,
+ /// Era layout, layer 8: which aligned window, below `2^win`. 0 on every op of every other class.
+ pub off: u8,
+}
+
+#[derive(Clone, Debug, PartialEq, Eq)]
+pub struct Program {
+ /// A label for packs and logs: the seed string, or whatever the caller named a byte seed.
+ pub seed_string: String,
+ /// The program seed bytes (`program_seed` of spec 01 section 1.12): the UTF-8 of a string seed, the 32-byte
+ /// epoch seed on the chain. Attempt `k` of this seed is `seed_words_from_bytes(seed_bytes || k_le32)`.
+ pub seed_bytes: Vec,
+ /// The seed words of this attempt (what the program stream and the register init use).
+ pub seed: [u32; 8],
+ /// Generator version, [`GENERATOR_VERSION`] for every current program; 1 for the retired generator.
+ pub generator: u32,
+ /// Attempt index: 0 for the bare seed, `k` for the k-th re-derivation after rejections.
+ pub attempt: u32,
+ /// The load class: [`LoadClass::V2`] for the lottery hash, another for the read-width experiment.
+ pub class: LoadClass,
+ /// The era seed bytes a class v3 chain program was drawn under (`E_n` of spec 04 section 4.4, the devnet stand-in
+ /// of `docs/plans/era-layout.md` section 2), recorded in the pack so a worker can check it carries the era the
+ /// job names. `None` for every version 2 program and every string-seed pack. The placeholder [`V3_CLASS`] does
+ /// not read it; the era draw of the integration branch will.
+ pub era_bytes: Option>,
+ pub instrs: Vec,
+ /// The latency-shadow block (Counter ASIC 3.0 item 8): `class.shadow.instrs` ALU instructions, run
+ /// `class.shadow.reps` times at the end of every iteration. Empty for every class without a shadow.
+ pub shadow: Vec,
+}
+
+/// The widths a `load` may read, in words: 4, 16 and 64 bytes.
+pub const WIDTH_WORDS: [u8; 3] = [1, 4, 16];
+
+/// The load class of a program (read-width experiment, 5 October 2026). `mix` holds the percent weights of the
+/// three widths of [`WIDTH_WORDS`] (sum 100); `load_slots` the number of `load` instructions per program.
+#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
+pub struct LoadClass {
+ pub mix: [u8; 3],
+ pub load_slots: u8,
+ /// Variant 5: `Some(k)` gives the program a per-warp scratch (the kernels run persistent warps) and turns `k`
+ /// of the load slots into scratch read-modify-writes. `None` for every other class.
+ pub scratch: Option,
+ /// Variant 5: the scratch per warp in KiB (32 or 128; the whole working set of a card at full occupancy must
+ /// stay under 6 GB, coordinator's cap of 5 October 2026). 0 for every other class.
+ pub scratch_kb: u8,
+ /// Mixer cost multiplier `m` of the dataset item derivation (Counter ASIC 2.0, M16, decided 5 October 2026 for
+ /// class v3): every mixer application of spec 01 section 1.8.5 becomes `m` applications with distinct round
+ /// keys, the 8 dependent cache reads per item unchanged (`memhard::derive_items`). 1 for version 2, 4 for v3.
+ pub mixer_mult: u8,
+ /// Cache growth rule, option C (`memhard::growth_doublings`): the cache doubles when the dataset doubles. `false`
+ /// for version 2 (the cache is 2^26 words on every day), `true` for v3.
+ pub growth: bool,
+ /// Era layout (`docs/plans/era-layout.md`): `Some` turns on the strided, windowed load address and the
+ /// interleaved dataset mapping with the parameters drawn from the era seed. `None` for every other class.
+ pub era: Option,
+ /// Hot table (`docs/plans/hot-table.md`, measured 5 October 2026 and not adopted): `Some(HotClass { mb, k, added })`
+ /// turns `k` load slots into reads of an `mb` MiB epoch table. `None` for every other class, class v3 included.
+ pub hot: Option,
+ /// Counter ASIC 3.0 item 2 (`crate::derive`, `docs/plans/counter-asic-3-derivation.md`, 6 October 2026, a
+ /// prototype behind the class): instructions per round program of the per-day item-derivation program that
+ /// replaces the fixed mixer when non-zero (the mixer multiplier is then unused and 1). 0 for every other class.
+ pub derive_len: u16,
+ /// Latency-shadow program work (Counter ASIC 3.0 item 8, measured 6 October 2026 and not adopted): `Some` adds a
+ /// block of ALU instructions run `reps` times per iteration. `None` for every other class, class v3 included.
+ pub shadow: Option,
+ /// Class v5, proof of stored state and of following (`docs/design/class-v5-stored-state.md`, 7 October 2026):
+ /// the item derivation XORs the window's state leaf into every item before the first mixer (`crate::state`,
+ /// `memhard::derive_items_leaves`). The program draw does not read it. `false` for every other class.
+ pub state: bool,
+}
+
+/// The parameters one era draws from its seed `E_n` (`docs/plans/era-layout.md` section 1.1, the proposed text of
+/// spec 01 section 1.13.1). `Copy` so the class stays `Copy`; the eight words of the era stream's seed and the era
+/// index are carried so a pack can say where the draw came from.
+#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
+pub struct EraParams {
+ /// `seed_words_from_bytes("igneum-era/" || E_n)` (`E_n` commits to the era index through the VDF input of spec
+ /// 04 section 4.4 step 2, so the index does not enter the draw).
+ pub words: [u32; 8],
+ /// The genesis-fixed set the width is drawn from, ascending, zero-padded (`[1, 0, 0]` pins 4 bytes: the
+ /// read-width decision of 5 October 2026, v2's 128 x 4 B stays).
+ pub allowed: [u8; 3],
+ /// Layer 4, item size: the words one load folds (1, 4 or 16), drawn from the allowed set.
+ pub width_words: u8,
+ /// Layer 4, stride: `y = rotl(x * stride_mul, stride_rot)`; the multiplier is odd, the rotation in 1..31.
+ pub stride_mul: u32,
+ pub stride_rot: u32,
+ /// Layer 4, interleave: the four ascending bit positions (0..15) of the word-within-item bits in the word
+ /// index; the first `log2(width_words)` are `0..`, so one aligned load stays inside one item.
+ pub pos: [u8; 4],
+}
+
+/// Domain tag of the era stream seed.
+pub const ERA_TAG: &[u8] = b"igneum-era/";
+
+impl EraParams {
+ /// `seed_words_from_bytes("igneum-era/" || era_bytes)`.
+ pub fn stream_words(era_bytes: &[u8]) -> [u32; 8] {
+ let mut b = Vec::with_capacity(ERA_TAG.len() + era_bytes.len());
+ b.extend_from_slice(ERA_TAG);
+ b.extend_from_slice(era_bytes);
+ seed_words_from_bytes(&b)
+ }
+
+ /// The short label of the era in class names and pack lines: the first stream word as hex.
+ pub fn label(&self) -> String {
+ format!("{:08x}", self.words[0])
+ }
+
+ /// The era bytes of a test seed string: the 32 bytes (little-endian words) of `seed_words_from_bytes(s)`.
+ pub fn test_era_bytes(s: &str) -> [u8; 32] {
+ let w = seed_words_from_bytes(s.as_bytes());
+ let mut out = [0u8; 32];
+ for (i, x) in w.iter().enumerate() {
+ out[i * 4..i * 4 + 4].copy_from_slice(&x.to_le_bytes());
+ }
+ out
+ }
+
+ /// The dataset layout this era's loads and build use.
+ pub fn layout(&self) -> crate::memhard::Layout {
+ crate::memhard::Layout { pos: self.pos }
+ }
+
+ /// The allowed widths as a slice (the non-zero entries).
+ pub fn allowed_set(&self) -> Vec {
+ self.allowed.iter().copied().filter(|&w| w != 0).collect()
+ }
+
+ /// The bytes that enter the program id after `"era/"`: the allowed set, width, multiplier, rotation, positions.
+ pub fn id_bytes(&self) -> Vec {
+ let mut b = Vec::with_capacity(3 + 1 + 4 + 4 + 4);
+ b.extend_from_slice(&self.allowed);
+ b.push(self.width_words);
+ b.extend_from_slice(&self.stride_mul.to_le_bytes());
+ b.extend_from_slice(&self.stride_rot.to_le_bytes());
+ b.extend_from_slice(&self.pos);
+ b
+ }
+}
+
+/// The era draw (`docs/plans/era-layout.md` section 1.1): seven draws from one SplitMix64 stream seeded with words 0
+/// and 1 of [`EraParams::stream_words`], in this order: the width from `allowed` (ascending, a non-empty subset of
+/// [`WIDTH_WORDS`]; one element pins it, the draw is still consumed), the odd stride multiplier, the stride rotation
+/// in 1..31, then four draws for the interleave (a partial Fisher-Yates over the candidate positions `log2(W)..15`,
+/// `4 - log2(W)` of them used, the rest consumed).
+pub fn era_draw(era_bytes: &[u8], allowed: &[u8]) -> EraParams {
+ assert!(!allowed.is_empty() && allowed.len() <= 3, "the allowed width set has 1 to 3 entries");
+ for (i, &w) in allowed.iter().enumerate() {
+ assert!(WIDTH_WORDS.contains(&w), "allowed width {w} is not 1, 4 or 16 words");
+ assert!(i == 0 || allowed[i - 1] < w, "the allowed width set is ascending");
+ }
+ let words = EraParams::stream_words(era_bytes);
+ let mut s = SplitMix64::new(words[0] as u64 | ((words[1] as u64) << 32));
+ let width_words = allowed[s.below(allowed.len() as u64) as usize];
+ let stride_mul = (s.next() as u32) | 1;
+ let stride_rot = 1 + s.below(31) as u32;
+ let b = width_words.trailing_zeros() as usize; // 0, 2 or 4
+ let free = 4 - b;
+ let mut c: Vec = (b as u8..16).collect();
+ let mut r = [0u64; 4];
+ for x in r.iter_mut() {
+ *x = s.next();
+ }
+ for i in 0..free {
+ let n = c.len() - i;
+ let j = i + (r[i] % n as u64) as usize;
+ c.swap(i, j);
+ }
+ // Draws 8 and 9, consumed and not used (spec 01 section 1.13.1 for `epoch_len`; `docs/design/latency-ladder.md`
+ // section 2 for the latency ladder): both parameters are set by miner signal, and consuming their slots here means a
+ // later use of either changes no other draw. Nothing below reads them, so every value drawn above is what it was.
+ let _epoch_len_draw = s.next();
+ let _latency_ladder_draw = s.next();
+ let mut chosen: Vec = c[..free].to_vec();
+ chosen.sort_unstable();
+ let mut pos = [0u8; 4];
+ for i in 0..b {
+ pos[i] = i as u8;
+ }
+ for (i, p) in chosen.iter().enumerate() {
+ pos[b + i] = *p;
+ }
+ let mut al = [0u8; 3];
+ al[..allowed.len()].copy_from_slice(allowed);
+ EraParams { words, allowed: al, width_words, stride_mul, stride_rot, pos }
+}
+
+/// The hot table of a class: `mb` MiB (32, 64 or 96 in the experiment) and `k` hot slots. Two forms: `replaced`
+/// (`added = false`): `k` of the 16 load slots read the table, 16 - k dataset loads; `added` (`added = true`,
+/// coordinator's form of 5 October 2026 against the on-die-cache recompute chip): the program has 16 + k load slots,
+/// the `k` hot ones drawn among them, so the 16 dataset loads and the 4,096-item verifier bound are unchanged and the
+/// hot loads are extra work (a cache hit on a GPU, SRAM and a read on a chip).
+#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
+pub struct HotClass {
+ pub mb: u8,
+ pub k: u8,
+ pub added: bool,
+}
+
+/// Program work in the latency shadow (Counter ASIC 3.0 item 8, 6 October 2026, `docs/analysis/latency-shadow-2026-10-06.md`;
+/// NOT the lottery hash, a class v4 candidate's knob): a shadow block of `instrs` ALU instructions (the ten non-load
+/// families at the weights of [`NONLOAD_WEIGHTS`], the same nine draws per instruction as the program's, drawn from the
+/// program stream AFTER the 64 base instructions, so the base program, its attempt and its acceptance verdict are those
+/// of the class without the shadow, draw for draw) executed `reps` times at the end of every iteration, after instruction
+/// 63 and before the next iteration samples `sel`. The 16 loads per program, the fresh-source rule and the acceptance
+/// rule (which interprets the base program only) are untouched; the shadow adds `8 x instrs x reps` ALU instructions per
+/// hash and no load. `None` on every other class: version 2 and class v3 draw nothing and emit nothing.
+#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
+pub struct ShadowClass {
+ /// Instructions in the shadow block (1..=4096).
+ pub instrs: u16,
+ /// Times the block runs per iteration (1..=1024).
+ pub reps: u16,
+}
+
+impl ShadowClass {
+ /// Shadow instructions per hash: `ITERATIONS x instrs x reps`.
+ pub fn instrs_per_hash(&self) -> usize {
+ ITERATIONS * self.instrs as usize * self.reps as usize
+ }
+}
+
+/// Scratch geometry (variant 5): 16-byte slots, lane-major, 32 lanes per warp; `scratch_kb` KiB per warp gives
+/// `scratch_kb x 2` slots per lane (32 KiB: 64 slots, 128 KiB: 256 slots).
+pub const SCRATCH_SLOT_BYTES: usize = 16;
+
+impl LoadClass {
+ /// Slots per lane of the scratch (0 without one).
+ pub fn scratch_slots_per_lane(&self) -> usize {
+ self.scratch_kb as usize * 1024 / LANES / SCRATCH_SLOT_BYTES
+ }
+ pub fn scratch_slot_mask(&self) -> u32 {
+ self.scratch_slots_per_lane().saturating_sub(1) as u32
+ }
+ pub fn scratch_words_per_lane(&self) -> usize {
+ self.scratch_slots_per_lane() * 4
+ }
+ pub fn scratch_bytes_per_warp(&self) -> usize {
+ self.scratch_kb as usize * 1024
+ }
+}
+
+impl LoadClass {
+ /// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash.
+ pub const V2: LoadClass =
+ LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None, state: false };
+
+ /// The construction decided for program class v3 on 5 October 2026 (Counter ASIC 2.0, `docs/plans/mixer-x4.md`):
+ /// version 2 loads (16 slots of one word, no scratch, no width roll, so the program stream is version 2's), the
+ /// mixer applied 4 times per round, and the cache growth rule. Name "mx4".
+ pub const MX4: LoadClass =
+ LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None, state: false };
+
+ /// The era class over `base` (`docs/plans/era-layout.md`): the parameters drawn by [`era_draw`]; when `allowed`
+ /// has more than one width the drawn width becomes the class mix (every load that width), otherwise the base
+ /// class's mix stands (the width rule of the read-width branch, pinned) and the era's `width_words` is the widest
+ /// width that mix can draw.
+ pub fn era(base: LoadClass, era_bytes: &[u8], allowed: &[u8]) -> LoadClass {
+ let mut e = era_draw(era_bytes, allowed);
+ let mut c = base;
+ if allowed.len() > 1 {
+ let i = WIDTH_WORDS.iter().position(|&w| w == e.width_words).unwrap();
+ c.mix = [0, 0, 0];
+ c.mix[i] = 100;
+ } else {
+ let widest = (0..3).rev().find(|&i| c.mix[i] > 0).map(|i| WIDTH_WORDS[i]).unwrap_or(1);
+ if widest != e.width_words {
+ // the interleave must keep the widest load inside one item: redraw the positions for that width
+ e = era_draw(era_bytes, &[widest]);
+ }
+ }
+ c.era = Some(e);
+ c
+ }
+
+ /// The dataset layout of this class ([`crate::memhard::Layout::LINEAR`] without an era).
+ pub fn layout(&self) -> crate::memhard::Layout {
+ self.era.map(|e| e.layout()).unwrap_or(crate::memhard::Layout::LINEAR)
+ }
+
+ /// The x8 candidate beside [`LoadClass::MX4`] (coordinator's rule of 5 October 2026, 21:30 UTC: x8 enters v3 if
+ /// the per-warp verify stays under 10 ms on one Mac core and the daily 1 GiB build under 1 s on every card):
+ /// the same loads and growth rule, the mixer applied 8 times per round. Name "mx8".
+ pub const MX8: LoadClass =
+ LoadClass { mixer_mult: 8, ..LoadClass::MX4 };
+
+ /// Counter ASIC 3.0 item 2 (6 October 2026, `docs/plans/counter-asic-3-derivation.md`): version 2 loads and the
+ /// growth rule of class v3, with the per-day derivation program of `crate::derive` (736 instructions per round
+ /// program, the x8-equivalent operation count) in place of the mixer. Name "dr736". A prototype class; not v3.
+ pub const DR736: LoadClass =
+ LoadClass { mixer_mult: 1, derive_len: crate::derive::DERIVE_LEN_X8 as u16, ..LoadClass::MX4 };
+
+ /// This class with a derivation program of `len` instructions per round program (0: the fixed mixer); the
+ /// mixer multiplier is set to 1 under a program, since no mixer is applied.
+ pub fn with_derive(self, len: u16) -> LoadClass {
+ assert!(len == 0 || (32..=4096).contains(&len), "derivation program length must be 0 or 32..=4096");
+ LoadClass { derive_len: len, mixer_mult: if len != 0 { 1 } else { self.mixer_mult }, ..self }
+ }
+
+ /// Whether the item derivation is the per-day program (Counter ASIC 3.0 item 2).
+ pub fn is_derived(&self) -> bool {
+ self.derive_len != 0
+ }
+
+ /// A fixed width (1, 4 or 16 words) with `load_slots` loads per program.
+ pub fn fixed(width_words: u8, load_slots: u8) -> LoadClass {
+ let mut mix = [0u8; 3];
+ let i = WIDTH_WORDS.iter().position(|&w| w == width_words).expect("width must be 1, 4 or 16 words");
+ mix[i] = 100;
+ LoadClass { mix, load_slots, ..LoadClass::V2 }
+ }
+
+ /// Per-load width drawn from `mix` (percent for 4, 16, 64 bytes), 16 loads per program.
+ pub fn mixed(mix: [u8; 3]) -> LoadClass {
+ assert_eq!(mix.iter().map(|&m| m as u32).sum::(), 100, "the mix must sum to 100");
+ LoadClass { mix, ..LoadClass::V2 }
+ }
+
+ /// Variant 5: version 2 widths, 16 memory operations of which `k` are scratch read-modify-writes into a
+ /// scratch of `kb` KiB per warp (a power of two, 1 to 128: at least one slot per lane, under the 6 GB cap).
+ pub fn scratch(k: u8, kb: u8) -> LoadClass {
+ assert!(k as usize <= LOAD_SLOTS);
+ assert!(kb.is_power_of_two() && kb <= 128, "scratch per warp must be a power of two up to 128 KiB");
+ LoadClass { scratch: Some(k), scratch_kb: kb, ..LoadClass::V2 }
+ }
+
+ /// Hot table, replaced form: version 2 widths, 16 load slots of which `k` read an `mb` MiB epoch table (`hot64k4`).
+ pub fn hot(mb: u8, k: u8) -> LoadClass {
+ LoadClass::V2.with_hot(mb, k)
+ }
+
+ /// Hot table, added form: version 2 widths, 16 + `k` load slots of which `k` read the table (`hot64k4a`), the 16
+ /// dataset loads unchanged.
+ pub fn hot_added(mb: u8, k: u8) -> LoadClass {
+ LoadClass::V2.with_hot_added(mb, k)
+ }
+
+ /// This class with a hot table in the replaced form (composes with a width mix or a scratch: the hot slots are
+ /// drawn after the scratch slots, and `scratch + hot` must fit the slot count).
+ pub fn with_hot(mut self, mb: u8, k: u8) -> LoadClass {
+ assert!(mb >= 1, "a hot table needs at least 1 MiB");
+ assert!(self.scratch_slots() + k as usize <= self.load_slots as usize, "scratch and hot slots exceed the load slots");
+ self.hot = Some(HotClass { mb, k, added: false });
+ self
+ }
+
+ /// This class with a hot table in the added form: `k` load slots are added to the class's and read the table.
+ pub fn with_hot_added(mut self, mb: u8, k: u8) -> LoadClass {
+ assert!(mb >= 1, "a hot table needs at least 1 MiB");
+ assert!((self.load_slots as usize) + (k as usize) < INSTR_COUNT, "added hot slots exceed the instruction count");
+ self.load_slots += k;
+ self.hot = Some(HotClass { mb, k, added: true });
+ self
+ }
+
+ /// Hot slots per program (0 without a hot table).
+ pub fn hot_slots(&self) -> usize {
+ self.hot.map(|h| h.k as usize).unwrap_or(0)
+ }
+
+ /// Hot slots that were added to the slot count (0 for the replaced form and without a hot table).
+ pub fn hot_added_slots(&self) -> usize {
+ self.hot.map(|h| if h.added { h.k as usize } else { 0 }).unwrap_or(0)
+ }
+
+ /// Load slots that read the dataset: the slot count less the scratch and hot slots.
+ pub fn dataset_slots(&self) -> usize {
+ self.load_slots as usize - self.scratch_slots() - self.hot_slots()
+ }
+
+ /// This class with the mixer multiplier `m` (1, 2, 4, 8 or 16) and the cache growth rule on or off.
+ /// The class with a latency-shadow block of `instrs` instructions run `reps` times per iteration ("mx8+sh256x13").
+ pub fn with_shadow(self, instrs: u16, reps: u16) -> LoadClass {
+ assert!((1..=4096).contains(&instrs) && (1..=1024).contains(&reps), "shadow block: 1..=4096 instructions, 1..=1024 reps");
+ LoadClass { shadow: Some(ShadowClass { instrs, reps }), ..self }
+ }
+
+ /// This class with another class's era draw (tests: a rung's class composed with the chain's era).
+ pub fn with_era_of(self, other: &LoadClass) -> LoadClass {
+ LoadClass { era: other.era, ..self }
+ }
+
+ /// The class with the state leaves of class v5 folded into every item ("mx8+sh256x27+state").
+ pub fn with_state(self) -> LoadClass {
+ LoadClass { state: true, ..self }
+ }
+
+ /// Shadow instructions per hash (0 without a shadow).
+ pub fn shadow_instrs_per_hash(&self) -> usize {
+ self.shadow.map(|s| s.instrs_per_hash()).unwrap_or(0)
+ }
+
+ pub fn with_mixer(self, mixer_mult: u8, growth: bool) -> LoadClass {
+ assert!(mixer_mult >= 1 && mixer_mult <= 16 && mixer_mult.is_power_of_two(), "mixer multiplier must be 1, 2, 4, 8 or 16");
+ LoadClass { mixer_mult, growth, ..self }
+ }
+
+ /// The mixer multiplier as the item derivation uses it.
+ pub fn mixer_mult(&self) -> u32 {
+ self.mixer_mult as u32
+ }
+
+ /// Whether the loads of this class are version 2's: 16 one-word loads, no scratch. Such a class takes no width
+ /// roll, so its program stream is the version 2 stream draw for draw (the mixer and the cache are properties of
+ /// the dataset, not of the program).
+ pub fn v2_loads(&self) -> bool {
+ self.mix == [100, 0, 0] && self.load_slots as usize == LOAD_SLOTS + self.hot_added_slots() && self.scratch.is_none()
+ }
+
+ /// Whether every instruction takes the tenth draw (the width roll): every class whose loads are not version 2's.
+ pub fn takes_width_roll(&self) -> bool {
+ !self.v2_loads()
+ }
+
+ /// Parse "p4,p16,p64" or one of the names of [`LoadClass::name`] ("scr4k32": 4 scratch ops, 32 KiB per warp;
+ /// "mx4": the v3 construction; a trailing "m" and "g" set the mixer multiplier and the growth rule on any
+ /// load class, "w16m4g" for example).
+ pub fn parse(s: &str) -> Option {
+ // "+state": the state leaves of class v5 over any class (the suffix is outermost)
+ if let Some(base) = s.strip_suffix("+state") {
+ return Some(LoadClass::parse(base)?.with_state());
+ }
+ // "+shx": the latency-shadow block over any class (Counter ASIC 3.0 item 8)
+ if let Some((base, sh)) = s.rsplit_once("+sh") {
+ let (instrs, reps) = sh.split_once('x')?;
+ let (instrs, reps): (u16, u16) = (instrs.parse().ok()?, reps.parse().ok()?);
+ if !(1..=4096).contains(&instrs) || !(1..=1024).contains(&reps) {
+ return None;
+ }
+ return Some(LoadClass::parse(base)?.with_shadow(instrs, reps));
+ }
+ if s == "mx4" {
+ return Some(LoadClass::MX4);
+ }
+ if s == "mx8" {
+ return Some(LoadClass::MX8);
+ }
+ // Counter ASIC 3.0 item 2: "dr" is the derivation class on the v3 loads and growth rule
+ if let Some(digits) = s.strip_prefix("dr") {
+ if !digits.is_empty() && digits.bytes().all(|b| b.is_ascii_digit()) {
+ let len: u16 = digits.parse().ok()?;
+ if len == 0 || !(32..=4096).contains(&len) {
+ return None;
+ }
+ return Some(LoadClass::MX4.with_derive(len));
+ }
+ }
+ // the mixer suffix: "...m" then an optional "g"
+ let (s, growth) = match s.strip_suffix('g') {
+ Some(base) if base.rsplit_once('m').map(|(_, d)| !d.is_empty() && d.bytes().all(|b| b.is_ascii_digit())).unwrap_or(false) => (base, true),
+ _ => (s, false),
+ };
+ if let Some((base, digits)) = s.rsplit_once('m') {
+ if !digits.is_empty() && digits.bytes().all(|b| b.is_ascii_digit()) && !base.is_empty() && !base.ends_with(',') {
+ let mult: u8 = digits.parse().ok()?;
+ if mult == 0 || mult > 16 || !mult.is_power_of_two() {
+ return None;
+ }
+ return Some(LoadClass::parse_loads(base)?.with_mixer(mult, growth));
+ }
+ }
+ if growth {
+ return None;
+ }
+ LoadClass::parse_loads(s)
+ }
+
+ /// The load part of a class name (no mixer suffix).
+ fn parse_loads(s: &str) -> Option {
+ // "+hotk[a]" composes a hot table with any class; "hotk[a]" alone is the version 2 base;
+ // the "a" suffix is the added form (k slots added to the class's), without it the replaced form
+ if let Some((base, hot)) = s.split_once("+hot") {
+ let (added, hot) = match hot.strip_suffix('a') { Some(h) => (true, h), None => (false, hot) };
+ let (mb, k) = hot.split_once('k')?;
+ let (mb, k): (u8, u8) = (mb.parse().ok()?, k.parse().ok()?);
+ let c = LoadClass::parse_loads(base)?;
+ if mb == 0 || k == 0 {
+ return None;
+ }
+ if added {
+ if c.load_slots as usize + k as usize >= INSTR_COUNT {
+ return None;
+ }
+ return Some(c.with_hot_added(mb, k));
+ }
+ if c.scratch_slots() + k as usize > c.load_slots as usize {
+ return None;
+ }
+ return Some(c.with_hot(mb, k));
+ }
+ if let Some(rest) = s.strip_prefix("hot") {
+ let (added, rest) = match rest.strip_suffix('a') { Some(h) => (true, h), None => (false, rest) };
+ let (mb, k) = rest.split_once('k')?;
+ let (mb, k): (u8, u8) = (mb.parse().ok()?, k.parse().ok()?);
+ if mb == 0 || k == 0 || k as usize > LOAD_SLOTS {
+ return None;
+ }
+ return Some(if added { LoadClass::hot_added(mb, k) } else { LoadClass::hot(mb, k) });
+ }
+ if let Some(rest) = s.strip_prefix("scr") {
+ let (k, kb) = rest.split_once('k')?;
+ let k: u8 = k.parse().ok()?;
+ let kb: u8 = kb.parse().ok()?;
+ if k as usize > LOAD_SLOTS || !kb.is_power_of_two() || kb > 128 {
+ return None;
+ }
+ return Some(LoadClass::scratch(k, kb));
+ }
+ let (mix_s, slots) = match s.split_once("x") {
+ Some((m, n)) if !m.contains(',') => (m, n.parse::().ok()?),
+ _ => (s, LOAD_SLOTS as u8),
+ };
+ let mix: [u8; 3] = match mix_s {
+ "v2" => return Some(LoadClass::V2),
+ "w4" => [100, 0, 0],
+ "w16" => [0, 100, 0],
+ "w64" => [0, 0, 100],
+ m => {
+ let v: Vec = m.split(',').map(|x| x.trim().parse::().ok()).collect::