diff --git a/docs/analysis/cryptanalysis/report-acceptance-rule.md b/docs/analysis/cryptanalysis/report-acceptance-rule.md index 9b51b13ee..ca83cdc54 100644 --- a/docs/analysis/cryptanalysis/report-acceptance-rule.md +++ b/docs/analysis/cryptanalysis/report-acceptance-rule.md @@ -255,6 +255,13 @@ it folds in act as fresh randomness on both sides: the closed form and the live the 50-program widening is running (gap-50.log). Also: the cheap 256-unit proxy did rank real tail programs, and 100064 sits 0.0032 above the 0.98 floor at 2^20, so the floor is live in the population tail, not idle. +### Rule change 19:55 BST (box-hours honesty) + +The bounded class became 88 cores per box in total across all lanes: no new sweep starts except under the +per-box lock `flock /srv/builds/_adv/locks/sweep.lock`, one sweep per box at a time; the running shards 00 +and 01 finish as they are; my adv-live confirmations and the row-90 census count against the ceiling, so +nothing new starts until they end. Shards 02 to 09, the class-v5 check and any re-run go through the lock. + ### Live hot-set census (RUNNING) - box 2, 19:5x BST: `adv-live census --programs 2..201 --nonces 1000000 --threads 32` (200 consecutive accepted diff --git a/tools/attack/adv-accept-v5/Cargo.lock b/tools/attack/adv-accept-v5/Cargo.lock new file mode 100644 index 000000000..dbcec7322 --- /dev/null +++ b/tools/attack/adv-accept-v5/Cargo.lock @@ -0,0 +1,112 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "adv-accept-v5" +version = "0.1.0" +dependencies = [ + "igneum-pow", +] + +[[package]] +name = "igneum-pow" +version = "0.2.0" +dependencies = [ + "serde_json", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "syn" +version = "3.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "unicode-ident" +version = "1.0.26" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/tools/attack/adv-accept-v5/Cargo.toml b/tools/attack/adv-accept-v5/Cargo.toml new file mode 100644 index 000000000..49f9e9561 --- /dev/null +++ b/tools/attack/adv-accept-v5/Cargo.toml @@ -0,0 +1,21 @@ +[package] +name = "adv-accept-v5" +version = "0.1.0" +edition = "2021" +description = "adv-accept lane: the f8 hot-set harness over class v5 (vendored igneum-pow of branch class-v5, the dataset keyed by the window's state leaves), to ask whether a class v4 finding still reads hot under class v5" +license = "MIT" +publish = false + +[[bin]] +name = "adv-live-v5" +path = "src/live.rs" + +[dependencies] +igneum-pow = { path = "igneum-pow" } + +[workspace] + +[profile.release] +opt-level = 3 +lto = true +codegen-units = 1 diff --git a/tools/attack/adv-accept-v5/igneum-pow/.gitignore b/tools/attack/adv-accept-v5/igneum-pow/.gitignore new file mode 100644 index 000000000..2f7896d1d --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/.gitignore @@ -0,0 +1 @@ +target/ diff --git a/tools/attack/adv-accept-v5/igneum-pow/Cargo.lock b/tools/attack/adv-accept-v5/igneum-pow/Cargo.lock new file mode 100644 index 000000000..82ec948df --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/Cargo.lock @@ -0,0 +1,105 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "igneum-pow" +version = "0.2.0" +dependencies = [ + "serde_json", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "syn" +version = "3.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "unicode-ident" +version = "1.0.26" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/tools/attack/adv-accept-v5/igneum-pow/Cargo.toml b/tools/attack/adv-accept-v5/igneum-pow/Cargo.toml new file mode 100644 index 000000000..a5f0f4b63 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/Cargo.toml @@ -0,0 +1,30 @@ +[package] +name = "igneum-pow" +version = "0.2.0" +edition = "2021" +description = "Igneum random-program GPU proof-of-work: seed, program generator (version 2: fixed load count, fresh sources, acceptance rule), memory-hard dataset, CPU warp verifier and kernel emitters; the source of every program pack" +license = "MIT" +publish = false + +[lib] +name = "igneum_pow" +path = "src/lib.rs" + +[[bin]] +name = "igneum-pow" +path = "src/main.rs" + +[dependencies] + +[dev-dependencies] +serde_json = "1" + +# The cache fill is 2^22 ChaCha12 blocks and the vector tests derive thousands of items. +# Unoptimised builds would make `cargo test` take minutes, so the dev profile is optimised too. +[profile.dev] +opt-level = 3 + +[profile.release] +opt-level = 3 +lto = true +codegen-units = 1 diff --git a/tools/attack/adv-accept-v5/igneum-pow/README.md b/tools/attack/adv-accept-v5/igneum-pow/README.md new file mode 100644 index 000000000..06507cca4 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/README.md @@ -0,0 +1,181 @@ +# igneum-pow + +The Igneum lottery hash in Rust: the generator, the acceptance rule, the memory-hard dataset, the CPU verifier and the +kernel emitters. This is the crate the rusty-kaspa fork calls (`docs/fork-map.md`, rows a1 to a3) and, since +4 October 2026, the source of every program pack in `proto-cuda/packs/`. No dependency outside the standard +library; `serde_json` is a dev-dependency for reading the packs in the tests. + +Dates: 3 October 2026 (crate, bit-exact with the Swift prototype), 4 October 2026 (generator version 2 and the +acceptance rule; every vector re-cut). Toolchain: rustc 1.99.0 via rustup (the Homebrew 1.69 on PATH is too old; +use `~/.cargo/bin/cargo`). Crate version 0.2.0. + +## Modules + +| Module | What it is | Swift namesake | +|---|---|---| +| `seed` | 32-byte seed words from bytes (FNV-1a 64, four salts, finalised); `seed_words_from_bytes` is the boundary where the chain feeds the epoch seed; SplitMix64 | `seedWordsBytes`, `SplitMix64` | +| `generator` | version 2: 16 load slots drawn first from instructions 1..63, fresh-source loads, the other 48 ops from the ten non-load weights; attempts `k = 0, 1, ...` of a seed; the program id. The retired version 1 generator stays as `generate_v1` for the census and the lever measurements | `generateProgramV2`, `candidateProgram`, `generateProgramV1` | +| `accept` | the acceptance rule of spec 01 section 1.4.6: two static tests and the 64-unit dynamic test on the seed-keyed closed-form dataset | `acceptProgram` | +| `memhard` | 256 MiB cache (2^16 chains of 64 ChaCha12 blocks), mixer parameters, 8-round item derivation with the 32 lanes interleaved, `MemhardCpu::fetch` | `cpuFillCache`, `MixParams`, `deriveItems`, `MemhardCPU` | +| `verify` | the 32-lane warp interpreter, `DatasetMode::{ClosedForm, MemoryHard}`, `Epoch`, `hash_warp`, `verify_block` | `cpuWarp`, `DatasetSource` | +| `emit` | Metal, CUDA and OpenCL source, program.h, memhard.h, vectors.h, program.json, vectors.json, the header-bound kernels, `export_pack` | `generateMSL`, `generateCUDA`, `generateOpenCL`, `exportPack` | +| `bind` | header binding (spec 01 section 1.6): init words from `"igneum-block/" \|\| H \|\| nonce_hi_le32`, bound hash API on `Epoch`, the 256-bit pow mapping, the interim day seed bytes | `blockInitWords` | + +## Generator version 2 (4 October 2026) + +Adopted from `docs/analysis/weak-program-census-2026-10-03.md` (ledger M5 and M6). Three parts, all in this crate +and mirrored in `proto-metal/main.swift` so the Metal worker derives the same program from the same seed: + +| Part | Rule | Where | +|---|---|---| +| G1, exact load count | 16 `load` instructions per program, a uniform 16-subset of slots 1..63 drawn first by partial Fisher-Yates over the program stream; the other 48 ops from `add 12, xor 10, mul 8, mad 8, shfl 8, rotl 7, sub 6, mulhi 6, rotr 6, or 4` (sum 75). 128 loads per hash, 4,096 items per unit | `generator::candidate_from_words` | +| G2, fresh source | a load's source is drawn from the registers other than `dst` written by an earlier instruction and not read by a load since, so no load repeats an earlier load's address in the hash | same | +| R, acceptance | (a) no load whose source is unwritten since the previous load from it, cyclically; (b) every register has an `add`, `sub`, `xor`, `mad`, `shfl` or `load` write; (c) 64 units at base nonces from `SplitMix64(FNV-1a-64("igneum-accept/" \|\| seed words LE))`, init words = seed words, closed-form dataset `dataset_elem(idx, S[0], S[1])` at 2^28 words: no constant register bit, no lane-constant load site in any unit, fewer than 164 saturated final values, every output bit within 136 of 1,024, distinct addresses above 245,760 over the 2,048 hashes | `accept::check` | +| Attempts | a rejected candidate is replaced by `seed_words_from_bytes(seed \|\| k_le32)` for `k = 1, 2, ...`; 32 consecutive rejections are a consensus fault (probability below 2^-136 at the measured 5 percent rate) | `generator::generate_from_seed_bytes` | +| Program id | `FNV-1a-64("igneum-program/" \|\| 2_le32 \|\| seed words LE \|\| attempt_le32)`, written into program.json and program.h with the generator version and the attempt, so a version 1 pack or another attempt can never pass for the current program | `Program::program_id` | + +Measured on this crate (`igneum-pow accept`): `igneum-genesis` attempt 0 accepted, program id `bcc1248b10cc90f2`, +op mix `load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1`, 128.000 distinct addresses per +hash; `igneum-hourly` attempt 0, id `a4c4d00961c855df`; seeds `igneum-census-2026-10-03/22`, `/37` and `/51` have +attempt 0 rejected ((b) r7 without an injecting write; (c) 119.74 distinct addresses; (b) r4) and attempt 1 +accepted (ids `22ed0609d079f4cf`, `947705cc4eb1df0a`, `9869afcc028bf9f1`); those three are the conformance vectors +for the attempt rule. The rule costs 1.3 to 3.4 ms per seed on one core. The 20,000-program census under this +generator is in `docs/bench-log.md` (4 October 2026 entry). + +## The API the fork calls + +```rust +use igneum_pow::{Epoch, DatasetMode}; + +// Once per epoch and day: derives the accepted program and fills the 256 MiB cache (about 0.2 s on one core). +let epoch = Epoch::memory_hard("igneum-genesis", "2026-10-03"); + +let h: u64 = epoch.hash(nonce); // one nonce (computes its aligned 32-nonce warp) +let w: [u64; 32] = epoch.hash_warp(base_nonce); // one warp +let ok: bool = epoch.verify_block(nonce, target_u64); + +// Miner programs for the epoch: the pack every worker compiles. +let pack = igneum_pow::emit::export_pack(&epoch, "2026-10-03", "igneum node"); +pack.write_to(std::path::Path::new("out"))?; // kernel.cu, kernel.cl, program.metal, memhard.h, ... +``` + +## The header-bound form (what the chain uses) + +The pack form above initialises the lane registers from the program's own seed words, so one nonce has one +hash per epoch whatever block is mined. On the chain the init words commit to the block (spec 01 section 1.6, +`src/bind.rs`): + +``` +H = header hash with the nonce field zeroed, every other field as mined + (rusty-kaspa hash_override_nonce_time(header, 0, header.timestamp)) +nonce = 64 bits; lane nonce n = low 32 bits; nonce_hi = high 32 bits +I = seed_words_from_bytes("igneum-block/" || H || nonce_hi_le32) (49 bytes in) +hash = interpret(program, I, n) (section 1.7 of the spec) +pow256 = hash in the top 64 bits, low 192 bits zero (little-endian bytes 24..32) +valid = pow256 <= target256, which is exactly hash <= target256 >> 192 +``` + +`H` keeps the timestamp (Kaspa zeroes it in the pre-PoW hash and absorbs it in cSHAKE afterwards; the lane hash has +no afterwards, so a nonce would otherwise be reusable across timestamps). The interim day seed is +`"igneum-day/" || day_le64` with `day = timestamp_ms / 86,400,000`. The epoch seed bytes are the 32 bytes of the +epoch block hash (devnet); the program is `generate_from_seed_bytes(epoch_seed)`, attempts included. + +```rust +use igneum_pow::{bind, Epoch}; + +let epoch = Epoch::from_seed_bytes(epoch_hash.as_bytes(), &bind::day_bytes(day), "label"); +let lane: u64 = epoch.hash_bound(&prehash, nonce); // one 64-bit nonce +let warp: [u64; 32] = epoch.hash_warp_bound(&prehash, nonce); // its aligned 32-nonce group +let init = bind::block_init_words(&prehash, nonce); // what a GPU kernel takes as its argument +let same = epoch.hash_warp_init(&init, nonce as u32 & !31); // == warp +let pow: [u8; 32] = epoch.pow_bound(&prehash, nonce); +let ok = epoch.verify_block_bound(&prehash, nonce, bind::target64_from_le256(&target_le)); +``` + +On the GPU the init words are a kernel argument: `igneum_hash_bound` in `program_bound.metal` takes +`constant uint* initw [[buffer(3)]]`, `kernel_bound.cu` takes `IgneumInitWords iw` by value, `kernel_bound.cl` an +`initw` buffer. All three differ from `igneum_hash` only in the kernel name, the argument and the eight init lines. + +Bound vectors (seed `igneum-genesis`, day `2026-10-03`, memory-hard, 2^28 words, generator v2; `igneum-pow hash-bound`): + +| H | nonce | hash_bound | +|---|---|---| +| 32 zero bytes | 0 | `746c567b090acf6a` | +| 32 zero bytes | 1 | `45a619f860880c73` | +| 32 zero bytes | 31 | `9aa495e43dedbfe6` | +| 32 zero bytes | 4096 | `2e6ffd7624d3cba2` | +| 32 zero bytes | 4294967296 (1 << 32) | `38a5cea1fb01431a` | +| bytes 00 01 02 .. 1f | 0 | `2a79c5e4797bf6aa` | +| bytes 00 01 02 .. 1f | 4294967301 ((1 << 32) + 5) | `a243e0c61aa1b82e` | +| bytes 00 01 02 .. 1f | 18446744073709551615 (u64::MAX) | `9c7bbfbd064fe1a4` | + +Init words for H = 32 zero bytes, nonce 0: `595a8f8a 37647e95 faadade1 cbbcf2a4 54f7cc13 f6851b5e 8c68ca04 7991ea9c` +(unchanged by v2: the binding does not depend on the program). The eight are pinned in `bind::tests::bound_vectors`. +The version 1 bound vectors of 3 October 2026 are retired. + +`Epoch` is `Send + Sync`; build one and share it. `DatasetMode::ClosedForm` is the interpreter regression dataset +of the packs `igneum-genesis` and `igneum-hourly` and is not memory-hard. + +## CLI + +``` +cargo build --release +./target/release/igneum-pow bench --seed igneum-genesis [--warps 20] [--closed-form] [--day 2026-10-03] +./target/release/igneum-pow export --seed igneum-genesis --out [--closed-form] +./target/release/igneum-pow export --epoch-hex <64 hex> --day-hex --out the chain's byte seeds +./target/release/igneum-pow hash --seed igneum-genesis --nonce 4103 +./target/release/igneum-pow hash-bound --seed igneum-genesis --prehash <64 hex> --nonce +./target/release/igneum-pow accept --seed igneum-census-2026-10-03/22 every candidate with its verdict +./target/release/igneum-pow show --seed igneum-genesis the accepted program, one line per instruction +``` + +The packs were regenerated on 4 October 2026 with exactly these commands: + +``` +igneum-pow export --closed-form --seed igneum-genesis --out ../proto-cuda/packs/igneum-genesis +igneum-pow export --closed-form --seed igneum-hourly --out ../proto-cuda/packs/igneum-hourly +igneum-pow export --seed igneum-genesis --out ../proto-cuda/packs/igneum-genesis-mh +igneum-pow export --epoch-hex edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07 \ + --day-hex 69676e65756d2d6461792ffa50000000000000 --out ../proto-cuda/packs/igneum-devnet-v4-epoch0 +``` + +The last is the chain's own derivation for devnet v4 epoch 0: the devnet genesis hash as the epoch seed and +`bind::day_bytes(20730)` (2026-10-04) as the day bytes. + +## Tests + +`cargo test --release` (39 tests, about 2 s after compile; the dev profile is optimised so the cache fill is quick): + +| Check | Pack | Result | +|---|---|---| +| program.json instruction by instruction from `seed_bytes`, generator 2, attempt, program id, op mix, 128 loads, acceptance | all four | match | +| Mixer parameters (key, rot, mul, rc) | igneum-genesis-mh, igneum-devnet-v4-epoch0 | match | +| Cache head, last line, FNV-1a 64 (`48c4f5bf24166b2e` for day 2026-10-03, `448274a57f508cbc` for day bytes 20730) | the two memory-hard packs | match | +| Dataset head (16), `[MASK]`, 64 sampled words | all four | match | +| 96 hash vectors (3 warps x 32 lanes) | all four | 96/96 each | +| kernel.cu, kernel_bound.cu, program.metal, program_bound.metal, kernel.cl, kernel_bound.cl, program.h, program.json byte-identical; 16 masked loads per kernel | all four | identical | +| memhard.h, memhard.metal byte-identical | the two memory-hard packs | identical | +| vectors.json, vectors.h byte-identical; no stale file in any pack directory | all four | identical | +| bound vectors (8), bound warp == bound single, H and nonce_hi enter the hash | igneum-genesis-mh | pass | +| the devnet pack equals `Epoch::from_seed_bytes(genesis, day_bytes(20730))` | igneum-devnet-v4-epoch0 | pass | +| acceptance: instrumented interpreter == `hash_warp`, cyclic stale-load detection, injecting-write detection, rejection under 12.5 percent and distinct loads above 127 on 400 census seeds, version 1 programs mostly rejected | unit tests | pass | +| generator: 16 loads, none at instruction 0, contract on 200 candidates; fresh sources; attempt words; program ids separate versions and attempts | unit tests | pass | + +## Measured, Apple M5 Max, one core, release build + +| Step | 3 October 2026 (v1, 104 loads) | 4 October 2026 (v2, 128 loads, 4,096 items) | +|---|---|---| +| Cache fill, 256 MiB | 175 to 181 ms | 179 ms | +| CPU verify per warp, igneum-genesis, avg of 20 | 0.441 ms (3,328 items) | 0.631 ms | +| Cold single warps (bases 0, 4096, 1000000) | 0.41 to 0.87 ms | 0.67 to 0.81 ms | +| Acceptance rule per candidate | | 1.3 to 3.4 ms | +| Closed form, igneum-genesis | 0.002 ms | 0.002 ms | + +The 10 ms gate holds with a margin of about 16x steady on this core. Every verified unit now derives exactly 4,096 +items, the design bound of spec section 1.11. + +## Not done here + +- No GPU. The vectors tie this crate to the Metal, CUDA and OpenCL results through the packs; nothing here runs a kernel. +- The epoch seed enters at `Epoch::from_seed_bytes` (the epoch block hash on devnet); the VDF output replaces it there. +- The dynamic acceptance test is specified on the closed-form dataset at 2^28 words; if the prototype mask ever changes, the rule's constant stays at 2^28. diff --git a/tools/attack/adv-accept-v5/igneum-pow/VENDORED-FROM.txt b/tools/attack/adv-accept-v5/igneum-pow/VENDORED-FROM.txt new file mode 100644 index 000000000..76140bcb6 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/VENDORED-FROM.txt @@ -0,0 +1 @@ +class-v5 igneum-pow at commit 25c8063ff38e248b7a92d91575e975a81d953894 diff --git a/tools/attack/adv-accept-v5/igneum-pow/rustfmt.toml b/tools/attack/adv-accept-v5/igneum-pow/rustfmt.toml new file mode 100644 index 000000000..c775577ee --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/rustfmt.toml @@ -0,0 +1,2 @@ +max_width = 120 +use_small_heuristics = "Max" diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/accept.rs b/tools/attack/adv-accept-v5/igneum-pow/src/accept.rs new file mode 100644 index 000000000..709d6f432 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/accept.rs @@ -0,0 +1,869 @@ +//! Program acceptance (spec 01 section 1.4.6, adopted 4 October 2026 from the weak-program census of +//! `docs/analysis/weak-program-census-2026-10-03.md`, section 6). +//! +//! A candidate program is accepted only if every test below holds. Every conforming implementation evaluates +//! exactly these tests on exactly these inputs, so every node skips the same seeds. +//! +//! | Part | Test | +//! |---|---| +//! | (a) static | for every `load`, some instruction between the previous `load` from the same source register and this one, in cyclic order over the 64 instructions, writes that register | +//! | (b) static | every register `r0..r7` is the destination of at least one `add`, `sub`, `xor`, `mad`, `shfl` or `load` | +//! | (c) dynamic | the program is interpreted for [`ACCEPT_UNITS`] (64) units of 32 lanes at base nonces drawn from SplitMix64 seeded with `FNV-1a-64("igneum-accept/" \|\| seed words as little-endian bytes)`, each `low32(next()) AND NOT 31`, with init words equal to the seed words and the closed-form dataset `dataset_elem(idx, S[0], S[1])` at [`ACCEPT_DATASET_LOG2`] (2^28 words) in place of the memory-hard dataset. Over the 2,048 evaluations: no register has a bit equal in every final value; no load site (iteration, instruction) reads one address in all 32 lanes of any unit; fewer than [`MAX_SATURATED`] (164, 1 percent of 16,384) final register values are 0 or 2^32 - 1; every output bit's ones count is within [`BIAS_TOLERANCE`] (136, 6 sigma) of 1,024; the distinct masked addresses read by one lane in one evaluation, summed over the 2,048 evaluations, exceed [`MIN_DISTINCT_SUM`] (245,760, a mean above 120 of the 128 loads) | +//! +//! The dynamic test uses the closed form so that it is a pure function of the program (no cache, no day) and +//! costs about a millisecond on one core. A hot-table load (`docs/plans/hot-table.md`) reads the closed form keyed by +//! seed words 2 and 3 at its multiply-shift index, a second pure table beside the dataset stand-in (words 0 and 1). The census (section 7.3) checked on 100,000 programs that the +//! closed-form verdict agrees with the memory-hard one on all but 39 threshold-edge cases. + +use crate::generator::{Instr, LoadClass, Op, Program, ShadowClass, INSTR_COUNT, ITERATIONS, LANES, V4_CLASS, V4_SHADOW_INSTRS}; +use crate::seed::{fnv1a64, SplitMix64}; +use crate::memhard::hot_index; +use crate::verify::{dataset_elem, fold_words, load_index, splitmix32, ScratchModel}; + +/// Units (32-lane warps) the dynamic test interprets. +pub const ACCEPT_UNITS: usize = 64; + +/// Class v4 sub-version 3, rule (c''): the per-site distinct-index RATIO (AP-F8-1's low-entropy-band class, 7 October +/// 2026, the attack-pass gate's numbers through main), keyed on the class v4 shape. Over [`ACCEPT_UNITS_DISTINCT_V4`] +/// units (2^20 evaluations per site) the count of distinct dataset word indices a load site reads, against the +/// expectation of a uniform source on the site's window (N - N^2 / 2W), must reach [`MIN_DISTINCT_RATIO_V4`]. The +/// floor sits between the 55 clean F8 seeds' minimum over their site rows (0.9960; the clean p1 0.9990, the median +/// 1.0000) and the strong failing seeds' maximum (p56 0.9654; p23 0.8361, p18 0.9274, p19 0.9335, p15 0.9432), +/// 0.015 from each. The open tail: F8's p4, p8, p10 and p34 (1.22x to 1.50x on the gate) read 0.9927 to 0.9963 at +/// 2^20, inside the clean spread, and a 2^24 pass does not separate them either (p34 0.9181, p4 0.9614, p8 0.9630, +/// p10 0.9612 against the clean p44 0.9612, p52 0.9613, p3 0.9971, p2 and p5 1.0004); they stay unattributed and +/// chased in `docs/fud-ledger.md` AP-F8-1. A candidate under the floor is rejected and the next attempt drawn under +/// the 256 cap and the last resort. Measured on one box-2 core with the shadow executed: 2.8 s per chosen candidate. +pub const ACCEPT_UNITS_DISTINCT_V4: usize = 4096; +/// The ratio floor at 2^20. +pub const MIN_DISTINCT_RATIO_V4: f64 = 0.98; +/// Kept for the record and the driver, not wired: the most repeated source value per site over the (c) units' +/// 16,384 evaluations (a uniform site repeats a value 2 or 3 times; the finding's bands sit under the ratio instead). +pub const MAX_SOURCE_REPEAT_V4: u32 = 8; +/// Hashes the dynamic test evaluates: 2,048. +pub const ACCEPT_HASHES: usize = ACCEPT_UNITS * LANES; +/// Domain tag of the base-nonce stream. +pub const ACCEPT_TAG: &[u8] = b"igneum-accept/"; +/// log2 of the closed-form dataset the test addresses: the prototype's 2^28 words, MASK 0x0fffffff. +pub const ACCEPT_DATASET_LOG2: u32 = 28; +/// Final register values equal to 0 or 2^32 - 1 must number fewer than this (1 percent of 8 x 2,048). +pub const MAX_SATURATED: u32 = 164; +/// Every output bit's ones count must be within this of 1,024 (6 x sqrt(2048) / 2, rounded). +pub const BIAS_TOLERANCE: u32 = 136; +/// Distinct addresses per lane per evaluation, summed over 2,048 evaluations, must exceed this (mean above 120). +pub const MIN_DISTINCT_SUM: u64 = 245_760; + +/// The distinct-address bound for a program with `loads` dataset loads per hash: the same 120 of 128 ratio, so +/// [`MIN_DISTINCT_SUM`] for the lottery hash and `loads x 1,920` for the read-width classes with other counts. +/// Variant 5's scratch read-modify-writes are not dataset loads: their slots repeat by design (a later +/// read-modify-write sees an earlier write), so they are neither counted nor bounded here. +pub fn min_distinct_sum(loads: usize) -> u64 { + loads as u64 * ACCEPT_HASHES as u64 * 120 / 128 +} + +/// Why a candidate was rejected. The verdict (accept or reject) is what consensus depends on; the reason is the +/// first failing test in the order of the module table. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Reject { + /// (a): instruction `instr` loads from `reg`, which no instruction wrote since the previous load from it. + StaleLoadSource { instr: u8, reg: u8 }, + /// (b): no injecting op writes `reg`. + NoInjectingWrite { reg: u8 }, + /// (c): `reg` has `bits` bits equal in all 2,048 final values. + ConstantBit { reg: u8, bits: u8 }, + /// (c): the load at `instr` in `iteration` read one address in all 32 lanes of `unit`. + LaneConstantSite { iteration: u8, instr: u8, unit: u8 }, + /// (c): `count` final register values were 0 or all ones. + Saturated { count: u32 }, + /// (a'), class v4 sub-version 2 (AP-F8-1): the load at `instr` reads `reg`, which is not fresh by dataflow in the + /// steady state of the loop (the freshness fixpoint over the base program and the shadow block). + UnfreshLoadSource { instr: u8, reg: u8 }, + /// (c'), class v4 sub-version 2 (AP-F8-1): the load at `site` read a source value of 0 or all ones in `count` of + /// its 16,384 evaluations (64 units x 32 lanes x 8 iterations); limit [`MAX_SATURATED`] - 1, the same 1 percent as (c). + SaturatedSource { site: u8, count: u32 }, + /// (c'') (B), class v4 sub-version 3: the load at `site` read the value `value` in `count` of its 16,384 (c) + /// evaluations (limit [`MAX_SOURCE_REPEAT_V4`] - 1): one constant upstream that the lineage rule cannot see. + RepeatedSource { site: u8, value: u32, count: u32 }, + /// (c''), class v4 sub-version 3: the load at `site` read `distinct` distinct dataset word indices over + /// `evaluations`, `ratio_milli` / 1000 of a uniform source on its window, under the floor: a low-entropy index band + /// (F8's p23, p18, p19, p15, p56). + LowEntropySite { site: u8, distinct: u32, evaluations: u32, ratio_milli: u32 }, + /// (c): output bit `bit` was set in `ones` of 2,048 hashes. + OutputBias { bit: u8, ones: u32 }, + /// (c): the distinct-address sum was `sum`. + DistinctAddresses { sum: u64 }, +} + +impl std::fmt::Display for Reject { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Reject::StaleLoadSource { instr, reg } => { + write!(f, "(a) load at instruction {instr} reads r{reg}, unwritten since the previous load from it") + } + Reject::NoInjectingWrite { reg } => write!(f, "(b) r{reg} has no add, sub, xor, mad, shfl or load write"), + Reject::ConstantBit { reg, bits } => write!(f, "(c) r{reg} has {bits} nonce-independent bits"), + Reject::LaneConstantSite { iteration, instr, unit } => { + write!(f, "(c) load at iteration {iteration} instruction {instr} reads one address in all lanes of unit {unit}") + } + Reject::Saturated { count } => write!(f, "(c) {count} of 16384 final register values saturated (limit 163)"), + Reject::UnfreshLoadSource { instr, reg } => write!(f, "(a') load at {instr} reads r{reg}, not fresh by dataflow in the loop's steady state (class v4 sub-version 2)"), + Reject::RepeatedSource { site, value, count } => write!(f, "(c'') load site {site} read the value {value:#010x} in {count} of 16384 evaluations (limit {})", MAX_SOURCE_REPEAT_V4 - 1), + Reject::LowEntropySite { site, distinct, evaluations, ratio_milli } => write!(f, "(c'') load site {site} read {distinct} distinct word indices over {evaluations} evaluations, {}.{:03} of a uniform source on its window (floor {MIN_DISTINCT_RATIO_V4} at 2^20)", ratio_milli / 1000, ratio_milli % 1000), + Reject::SaturatedSource { site, count } => write!(f, "(c') load site {site} read a saturated source value in {count} of 16384 evaluations (limit 163)"), + Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"), + Reject::DistinctAddresses { sum } => { + write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0) + } + } + } +} + +/// What the dynamic test measured on an accepted program. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct AcceptReport { + /// Distinct masked addresses per lane per evaluation, summed over the 2,048 evaluations. + pub distinct_sum: u64, + /// Final register values equal to 0 or all ones. + pub saturated: u32, + /// The largest `|ones - 1024|` over the 64 output bits. + pub bias_max: u32, +} + +impl AcceptReport { + /// Mean distinct addresses per hash (128 at most). + pub fn distinct_mean(&self) -> f64 { + self.distinct_sum as f64 / ACCEPT_HASHES as f64 + } +} + +/// Part (a): no load whose source is unwritten since the previous load from it, cyclically. +fn check_stale_loads(instrs: &[Instr]) -> Result<(), Reject> { + // `pending[r]`: a load has read r and nothing has written r since. Two passes over the list so the second + // pass sees the state carried over the iteration boundary. + let mut pending = [false; 8]; + for _pass in 0..2 { + for (k, ins) in instrs.iter().enumerate() { + if ins.op.is_load() && pending[ins.src as usize] { + return Err(Reject::StaleLoadSource { instr: k as u8, reg: ins.src }); + } + pending[ins.dst as usize] = false; + if ins.op.is_load() { + pending[ins.src as usize] = true; + } + } + } + Ok(()) +} + +/// Part (b): every register has an injecting write. +fn check_injecting_writes(instrs: &[Instr]) -> Result<(), Reject> { + let mut injected = [false; 8]; + for ins in instrs { + if ins.op.injects() { + injected[ins.dst as usize] = true; + } + } + for (reg, ok) in injected.iter().enumerate() { + if !ok { + return Err(Reject::NoInjectingWrite { reg: reg as u8 }); + } + } + Ok(()) +} + +/// The most repeated value of `values` (sorted in place) and that value: (B), kept for the driver, not wired. +#[allow(dead_code)] +fn most_repeated(values: &mut [u32]) -> (u32, u32) { + values.sort_unstable(); + let (mut best, mut best_v, mut run) = (0u32, 0u32, 0u32); + for i in 0..values.len() { + run = if i > 0 && values[i] == values[i - 1] { run + 1 } else { 1 }; + if run > best { + best = run; + best_v = values[i]; + } + } + (best, best_v) +} + +/// (c''), class v4 sub-version 3: the 2^20 ratio pass on the chosen candidate (the constants above). +pub fn check_distinct_indices_v4(p: &Program) -> Result<(), Reject> { + distinct_ratio_pass(p, ACCEPT_UNITS_DISTINCT_V4, MIN_DISTINCT_RATIO_V4).map(|_| ()) +} + +/// One ratio pass over `units`: every load site's distinct word indices against the uniform expectation on its +/// window (`N - N^2 / 2W`, the window `2^28 >> min(win, 2)` words of the closed-form dataset), `Err` at the first +/// site under `floor`, else the minimum ratio and its site. +pub fn distinct_ratio_pass(p: &Program, units: usize, floor: f64) -> Result<(f64, usize), Reject> { + let n = (units * LANES * ITERATIONS) as f64; + let d = distinct_indices_v4(p, units)?; + let mut min = (f64::MAX, 0usize); + let mut site = 0usize; + for i in &p.instrs { + if !i.op.is_load() { + continue; + } + let wsize = ((1u64 << ACCEPT_DATASET_LOG2) >> (i.win as u64).min(2)) as f64; + let ratio = d[site] as f64 / (n - n * n / (2.0 * wsize)); + if ratio < floor { + return Err(Reject::LowEntropySite { site: site as u8, distinct: d[site], evaluations: n as u32, ratio_milli: (ratio * 1000.0) as u32 }); + } + if ratio < min.0 { + min = (ratio, site); + } + site += 1; + } + Ok(min) +} + +/// The distinct dataset word indices every load site reads over `units` units of the seed's acceptance stream on +/// the closed-form words (the sample caps the count near `units x 32 x 8`, so a site's index entropy is read only +/// below about log2 of that). +pub fn distinct_indices_v4(p: &Program, units: usize) -> Result, Reject> { + let loads = p.loads_per_hash(); + let sites = loads / ITERATIONS; + let mut acc = Acc { + sources: None, + indices: Some(vec![Vec::with_capacity(units * LANES * ITERATIONS); sites]), + sat_source: vec![0; sites], + and_acc: [u32::MAX; 8], + or_acc: [0; 8], + saturated: 0, + bit_ones: [0; 64], + distinct_sum: 0, + }; + let mut lane_addrs = vec![0u32; LANES * loads]; + for (unit, &base) in accept_base_nonces_n(&p.seed, units).iter().enumerate() { + run_unit(p, unit, base, &mut acc, &mut lane_addrs)?; + } + let mut out = Vec::with_capacity(sites); + for ix in acc.indices.take().unwrap().iter_mut() { + ix.sort_unstable(); + ix.dedup(); + out.push(ix.len() as u32); + } + Ok(out) +} + +/// Whether `class` is the class v4 shape (the 256-instruction shadow block over the class v3 base, the pass count and +/// the era set aside): the shape the sub-version 2 rules (a') and (c') apply to, on every draw path. +pub fn is_class_v4_shape(class: &LoadClass) -> bool { + matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. })) + // class v5 (docs/design/class-v5-stored-state.md) is judged under the same rules: its state flag is set aside + && LoadClass { era: None, shadow: None, state: false, ..*class } == LoadClass { shadow: None, ..V4_CLASS } +} + +/// One pass of the dataflow freshness over the base program then the shadow block (the order of one iteration), +/// from `fresh`; `check` reports the first load that reads a register that is not fresh. The rule (AP-F8-1, +/// `docs/analysis/ca3-v4-uniform.md`): a load leaves its destination fresh only if its source was (a saturated +/// source reads one fixed word); add, sub, xor, mad and shfl if either operand was; rotl and rotr if the operand +/// was (a rotate maps all-ones and zero to themselves); or, mul and mulhi never. +fn freshness_pass(p: &Program, fresh: &mut [bool; 8], pair_op: &mut [Option<(Op, usize)>; 8], check: bool) -> Result<(), Reject> { + for (k, i) in p.instrs.iter().chain(p.shadow.iter()).enumerate() { + let (d, a) = (i.dst as usize, i.src as usize); + if check && i.op.is_load() && !fresh[a] { + return Err(Reject::UnfreshLoadSource { instr: k as u8, reg: i.src }); + } + // the shared-operand idiom (sub-version 3): or-then-xor or or-then-sub on one operand is `d & ~s`, xor-then-or + // is `d | s`: lossy, though the second op would inject on its own (F8's p23: `or r6 |= r4; xor r6 ^= r4`) + let masked = matches!((pair_op[d], i.op), (Some((Op::Or, s)), Op::Xor) | (Some((Op::Or, s)), Op::Sub) | (Some((Op::Xor, s)), Op::Or) if s == a); + fresh[d] = !masked + && match i.op { + Op::Load | Op::WLoad | Op::Scratch | Op::Hot => fresh[a], + Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl => fresh[d] || fresh[a], + Op::Rotl | Op::Rotr => fresh[d], + Op::Or | Op::Mul | Op::MulHi => false, + }; + pair_op[d] = if matches!(i.op, Op::Or | Op::Xor) && !masked { Some((i.op, a)) } else { None }; + for r in 0..8 { + if r != d { + if let Some((_, s)) = pair_op[r] { + if s == d { + pair_op[r] = None; + } + } + } + } + } + Ok(()) +} + +/// Part (a'), class v4 sub-version 2: every load's source is fresh by dataflow in the loop's steady state. The draw +/// of `candidate_from_words_class` keeps in-pass sources fresh; this closes the iteration boundary (a source last +/// written late in the previous iteration or in the shadow block, which the draw's no-eligible fallback can pick: +/// F8's p11, an `or` at 63 feeding a load at 1). The state starts all fresh (the init words are a per-lane hash of +/// the nonce) and is run to its fixpoint (it only ever falls, so at most 8 passes change it), then one checking pass. +pub fn check_fresh_sources_v4(p: &Program) -> Result<(), Reject> { + if !is_class_v4_shape(&p.class) { + return Ok(()); + } + let mut fresh = [true; 8]; + let mut pair_op: [Option<(Op, usize)>; 8] = [None; 8]; + for _ in 0..9 { + let before = (fresh, pair_op); + freshness_pass(p, &mut fresh, &mut pair_op, false)?; + if (fresh, pair_op) == before { + break; + } + } + freshness_pass(p, &mut fresh, &mut pair_op, true) +} + +/// Parts (a), (b) and, for class v4 sub-version 2, (a'). +pub fn check_static(p: &Program) -> Result<(), Reject> { + if p.instrs.len() != INSTR_COUNT { + panic!("acceptance needs a {INSTR_COUNT}-instruction program"); + } + check_stale_loads(&p.instrs)?; + check_injecting_writes(&p.instrs)?; + check_fresh_sources_v4(p) +} + +/// The 64 base nonces of the dynamic test for seed words `seed`. +pub fn accept_base_nonces(seed: &[u32; 8]) -> [u32; ACCEPT_UNITS] { + let v = accept_base_nonces_n(seed, ACCEPT_UNITS); + let mut out = [0u32; ACCEPT_UNITS]; + out.copy_from_slice(&v); + out +} + +/// The first `n` base nonces of the seed's acceptance stream (the (c) units are the first [`ACCEPT_UNITS`]). +pub fn accept_base_nonces_n(seed: &[u32; 8], n: usize) -> Vec { + let mut b = Vec::with_capacity(ACCEPT_TAG.len() + 32); + b.extend_from_slice(ACCEPT_TAG); + for w in seed { + b.extend_from_slice(&w.to_le_bytes()); + } + let mut rng = SplitMix64::new(fnv1a64(&b)); + (0..n).map(|_| (rng.next() as u32) & !31).collect() +} + +#[inline(always)] +fn mulhi32(a: u32, b: u32) -> u32 { + ((a as u64 * b as u64) >> 32) as u32 +} + +/// Accumulators of the dynamic test over the 64 units. +struct Acc { + /// (c'') (B): every load's source value per site, recorded when present. + sources: Option>>, + /// (c'') (A): every load's dataset index per site, recorded when present (the distinct-index pass only). + indices: Option>>, + /// (c'): per load site (the load's index within the iteration), how many of its evaluations read a source value + /// of 0 or all ones (class v4 sub-version 2; counted for every class, judged for class v4 only). + sat_source: Vec, + and_acc: [u32; 8], + or_acc: [u32; 8], + saturated: u32, + bit_ones: [u32; 64], + distinct_sum: u64, +} + +/// One unit of the dynamic test: the interpreter of `verify.rs` with the closed-form dataset, instrumented. +/// Returns the first lane-constant load site, if any. +fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut [u32]) -> Result<(), Reject> { + let seed = &p.seed; + let mask: u32 = (1u32 << ACCEPT_DATASET_LOG2) - 1; + let (d0, d1) = (seed[0], seed[1]); + let (h0, h1) = (seed[2], seed[3]); + let hot_words = p.hot_words(); + let loads = p.loads_per_hash(); + let mut r = [[0u32; LANES]; 8]; + for lane in 0..LANES { + let nonce = base.wrapping_add(lane as u32); + for i in 0..8 { + let mut x = nonce ^ seed[i]; + x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1)); + x = splitmix32(x); + r[i][lane] = x ^ seed[(i + 1) & 7]; + } + } + let mut idx = [0u32; LANES]; + let mut nload = 0usize; + let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None }; + let slot_mask = p.class.scratch_slot_mask(); + let era = p.class.era; + // Class v4 sub-version 3 (AP-F8-3, 7 October 2026): the acceptance interpreter runs the latency-shadow block + // after instruction 63 of every iteration, `reps` times with the iteration's `sel`, exactly as the hash does + // (verify.rs). Until this commit it ran the 64 base instructions only, so every dynamic test (c) judged a class v4 + // program the chain never hashes. The shadow block holds no load, so its instructions take the same arms. + let shadow_reps = p.shadow_reps(); + for it in 0..ITERATIONS { + let sel = r[0]; + let shadow_pass = (0..shadow_reps).flat_map(|_| p.shadow.iter().enumerate().map(|(k, i)| (INSTR_COUNT + k, i))); + for (k, ins) in p.instrs.iter().enumerate().chain(shadow_pass) { + let d = ins.dst as usize; + let a = ins.src as usize; + match ins.op { + Op::Scratch => { + // Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word). + let m = scratch.as_mut().expect("a scratch op needs a scratch class"); + for lane in 0..LANES { + idx[lane] = r[a][lane] & slot_mask; + } + if idx.iter().all(|&x| x == idx[0]) { + return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 }); + } + for lane in 0..LANES { + r[d][lane] = m.rmw(&p.seed, base, lane, idx[lane], r[d][lane]); + lane_addrs[lane * loads + nload] = 0x8000_0000 | idx[lane]; + } + nload += 1; + } + Op::Add => { + let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32); + let src = r[a]; + for lane in 0..LANES { + let s = (sel[lane] >> bit) & 1; + let c = if s != 0 { imm2 } else { imm }; + r[d][lane] = r[d][lane].wrapping_add(src[lane]).wrapping_add(c); + } + } + Op::Sub => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].wrapping_sub(src[lane]); + } + } + Op::Mul => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].wrapping_mul(src[lane]); + } + } + Op::MulHi => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = mulhi32(r[d][lane], src[lane]); + } + } + Op::Xor => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] ^= src[lane]; + } + } + Op::Or => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] |= src[lane]; + } + } + Op::Rotl => { + let n = ins.rot; + for lane in 0..LANES { + r[d][lane] = r[d][lane].rotate_left(n); + } + } + Op::Rotr => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].rotate_right(src[lane] & 31); + } + } + Op::Mad => { + let src = r[a]; + let src2 = r[ins.src2 as usize]; + for lane in 0..LANES { + r[d][lane] = src[lane].wrapping_mul(src2[lane]).wrapping_add(r[d][lane]); + } + } + Op::Shfl => { + let src = r[a]; + let m = ins.mask as usize; + for lane in 0..LANES { + r[d][lane] ^= src[lane ^ m]; + } + } + Op::Load => { + // Read-width experiment: a load of `width` words reads from the aligned address and folds every + // word (verify::fold_words); width 1 is the lottery hash's xor of one word. + let width = ins.width as usize; + let align = !(ins.width as u32 - 1); + let site = nload % (loads / ITERATIONS); + for lane in 0..LANES { + let v = r[a][lane]; + acc.sat_source[site] += (v == 0 || v == u32::MAX) as u32; + if let Some(src) = acc.sources.as_mut() { + src[site].push(v); + } + idx[lane] = load_index(era.as_ref(), ins, v, mask, ACCEPT_DATASET_LOG2) & align; + if let Some(ix) = acc.indices.as_mut() { + ix[site].push(idx[lane]); + } + } + if idx.iter().all(|&x| x == idx[0]) { + return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 }); + } + for lane in 0..LANES { + if width == 1 { + r[d][lane] ^= dataset_elem(idx[lane], d0, d1); + } else { + let mut w = [0u32; 16]; + for j in 0..width { + w[j] = dataset_elem(idx[lane] + j as u32, d0, d1); + } + r[d][lane] = fold_words(r[d][lane], &w[..width]); + } + lane_addrs[lane * loads + nload] = idx[lane]; + } + nload += 1; + } + Op::Hot => { + // Hot table: the stand-in is dataset_elem keyed by seed words 2 and 3; the address is tagged with + // bit 30 so a hot word and a dataset word at one index count as two addresses. + for lane in 0..LANES { + idx[lane] = hot_index(r[a][lane], hot_words); + } + if idx.iter().all(|&x| x == idx[0]) { + return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 }); + } + for lane in 0..LANES { + r[d][lane] ^= dataset_elem(idx[lane], h0, h1); + lane_addrs[lane * loads + nload] = 0x4000_0000 | idx[lane]; + } + nload += 1; + } + Op::WLoad => { + let b = (r[a][0] & mask) & !31; + for lane in 0..LANES { + idx[lane] = b + lane as u32; + r[d][lane] ^= dataset_elem(idx[lane], d0, d1); + lane_addrs[lane * loads + nload] = idx[lane]; + } + nload += 1; + } + } + } + } + for i in 0..8 { + for lane in 0..LANES { + let v = r[i][lane]; + acc.and_acc[i] &= v; + acc.or_acc[i] |= v; + acc.saturated += (v == 0 || v == u32::MAX) as u32; + } + } + for lane in 0..LANES { + let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21); + let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27); + let h = ((hi as u64) << 32) | lo as u64; + for j in 0..64 { + acc.bit_ones[j] += ((h >> j) & 1) as u32; + } + let sl = &mut lane_addrs[lane * loads..(lane + 1) * loads]; + sl.sort_unstable(); + let mut distinct = 0u64; + for k in 0..loads { + // scratch slots carry bit 31 (variant 5) and are not dataset addresses + if sl[k] & 0x8000_0000 == 0 && (k == 0 || sl[k] != sl[k - 1]) { + distinct += 1; + } + } + acc.distinct_sum += distinct; + } + Ok(()) +} + +/// Part (c). +pub fn check_dynamic(p: &Program) -> Result { + let loads = p.loads_per_hash(); + let v4 = is_class_v4_shape(&p.class); + let sites = loads / ITERATIONS; + let mut acc = Acc { + sources: None, + indices: None, + sat_source: vec![0; sites], + and_acc: [u32::MAX; 8], + or_acc: [0; 8], + saturated: 0, + bit_ones: [0; 64], + distinct_sum: 0, + }; + let mut lane_addrs = vec![0u32; LANES * loads]; + for (unit, &base) in accept_base_nonces(&p.seed).iter().enumerate() { + run_unit(p, unit, base, &mut acc, &mut lane_addrs)?; + } + for reg in 0..8 { + let bits = (acc.and_acc[reg] | !acc.or_acc[reg]).count_ones(); + if bits != 0 { + return Err(Reject::ConstantBit { reg: reg as u8, bits: bits as u8 }); + } + } + if acc.saturated >= MAX_SATURATED { + return Err(Reject::Saturated { count: acc.saturated }); + } + // (c'), class v4 sub-version 2 (AP-F8-1, 7 October 2026): a load whose source is saturated reads one fixed word, + // whatever delivered the saturation (an or-written value, a rotate of one, a load after a saturated load); the + // source rule of the draw removes the writers it can see and this count catches every delivery. Keyed on the + // class v4 shape as the draw's rule is, so v2 and v3 verdicts do not move. + if v4 { + if let Some((site, &count)) = acc.sat_source.iter().enumerate().find(|(_, &c)| c >= MAX_SATURATED) { + return Err(Reject::SaturatedSource { site: site as u8, count }); + } + // (c''), the ratio on the candidate that passed everything else (the draw's last and dearest test) + check_distinct_indices_v4(p)?; + } + let half = (ACCEPT_HASHES / 2) as u32; + let mut bias_max = 0u32; + for (bit, &ones) in acc.bit_ones.iter().enumerate() { + let d = ones.abs_diff(half); + if d > BIAS_TOLERANCE { + return Err(Reject::OutputBias { bit: bit as u8, ones }); + } + bias_max = bias_max.max(d); + } + if acc.distinct_sum <= min_distinct_sum(loads - p.scratch_ops_per_hash()) { + return Err(Reject::DistinctAddresses { sum: acc.distinct_sum }); + } + Ok(AcceptReport { distinct_sum: acc.distinct_sum, saturated: acc.saturated, bias_max }) +} + +/// The whole rule: (a), (b), then (c). +pub fn check(p: &Program) -> Result { + check_static(p)?; + check_dynamic(p) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::generator::{candidate, candidate_class, generate, generate_class, GeneratorConfig, generate_v1, LoadClass}; + use crate::verify::{DatasetMode, DatasetSource}; + + #[test] + fn distinct_bound_scales_with_the_load_count() { + assert_eq!(min_distinct_sum(128), MIN_DISTINCT_SUM); + assert_eq!(min_distinct_sum(32), 61_440); + } + + /// The read-width classes pass the rule at about the version 2 rate, and the instrumented interpreter agrees + /// with `verify.rs` on every class (the fold is shared, the addresses are aligned the same way). + #[test] + fn classes_pass_and_match_verify() { + for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2k32", "scr8k128"] { + let c = LoadClass::parse(name).unwrap(); + let p = generate_class("igneum-genesis", c); + assert!(check(&p).is_ok(), "{name}"); + let mut rejected = 0; + for i in 0..60u32 { + let s = format!("igneum-rw-accept/{i}"); + let q = candidate_class(&s, s.as_bytes(), 0, c); + if check(&q).is_err() { + rejected += 1; + } + } + assert!(rejected < 15, "{name}: {rejected} of 60 rejected"); + let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2); + let bases = accept_base_nonces(&p.seed); + let loads = p.loads_per_hash(); + let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 }; + let mut la = vec![0u32; LANES * loads]; + let mut ones = [0u32; 64]; + for (u, &b) in bases.iter().enumerate() { + run_unit(&p, u, b, &mut acc, &mut la).unwrap(); + for h in crate::verify::hash_warp(&p, b, &ds) { + for j in 0..64 { + ones[j] += ((h >> j) & 1) as u32; + } + } + } + assert_eq!(acc.bit_ones, ones, "{name}: bit counts match the reference interpreter"); + } + } + + /// AP-F8-3 (7 October 2026): the acceptance's execution and `verify.rs` agree on a class v4 program WITH its + /// shadow block (the output bit counts over the 64 units on the closed-form dataset, the same sel per iteration), + /// so the two paths cannot diverge again: until sub-version 3 the acceptance ran the base program only and judged + /// a program the chain never hashes. The devnet epoch-0 seed and the six test eras, 8 x 256 x 27 shadow + /// instructions per hash each; the same program with its shadow stripped gives other counts. + #[test] + fn acceptance_executes_the_shadow_block_as_the_verifier_does() { + use crate::generator::{generate_era, EraParams, V3_ALLOWED, V4_CLASS}; + let hx = |h: &str| -> Vec { (0..h.len()).step_by(2).map(|i| u8::from_str_radix(&h[i..i + 2], 16).unwrap()).collect() }; + let g = hx("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07"); + let mut eras = vec![g.clone()]; + for n in 0..6 { + eras.push(EraParams::test_era_bytes(&format!("igneum-era-test/{n}")).to_vec()); + } + for era in &eras { + let p = generate_era("igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07", &g, V4_CLASS, era, &V3_ALLOWED); + assert_eq!(p.shadow.len(), 256); + assert_eq!(p.shadow_reps(), 27); + let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2); + let bases = accept_base_nonces(&p.seed); + let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 }; + let mut la = vec![0u32; LANES * p.loads_per_hash()]; + let mut ones = [0u32; 64]; + for (u, &b) in bases.iter().enumerate() { + run_unit(&p, u, b, &mut acc, &mut la).unwrap(); + for h in crate::verify::hash_warp(&p, b, &ds) { + for j in 0..64 { + ones[j] += ((h >> j) & 1) as u32; + } + } + } + assert_eq!(acc.bit_ones, ones, "the acceptance's execution of a class v4 program (shadow block included) matches the verifier's hashes"); + // and the same program with its shadow stripped hashes differently: the shadow is executed, not skipped + let mut bare = p.clone(); + bare.shadow.clear(); + let mut acc2 = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 }; + for (u, &b) in bases.iter().enumerate() { + let _ = run_unit(&bare, u, b, &mut acc2, &mut la); + } + assert_ne!(acc.bit_ones, acc2.bit_ones, "the shadow block changes the acceptance's execution"); + } + } + + /// The instrumented interpreter agrees with `verify.rs` on the closed-form dataset keyed by the seed words. + #[test] + fn instrumented_interpreter_matches_verify() { + for i in 0..20u32 { + let s = format!("igneum-accept-test/{i}"); + let p = candidate(&s, s.as_bytes(), 0); + let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2); + let bases = accept_base_nonces(&p.seed); + let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 }; + let mut la = vec![0u32; LANES * p.loads_per_hash()]; + let mut ones = [0u32; 64]; + let mut any = false; + for (u, &b) in bases.iter().enumerate() { + if run_unit(&p, u, b, &mut acc, &mut la).is_err() { + continue; + } + any = true; + let w = crate::verify::hash_warp(&p, b, &ds); + for h in w { + for j in 0..64 { + ones[j] += ((h >> j) & 1) as u32; + } + } + } + if any { + assert_eq!(acc.bit_ones, ones, "bit counts of {s} match the reference interpreter's hashes"); + } + } + } + + /// Hot-table experiment: the hot classes pass the rule at about the version 2 rate, and the hot addresses are + /// uniform over the table (16 buckets of the index over 64 units x 32 lanes x 32 hot loads). + #[test] + fn hot_classes_pass_and_hot_loads_are_uniform() { + for name in ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "scr4k32+hot64k4", "hot32k4a", "hot64k4a", "hot96k4a"] { + let c = LoadClass::parse(name).unwrap(); + let p = generate_class("igneum-genesis", c); + assert!(check(&p).is_ok(), "{name}"); + let mut rejected = 0; + for i in 0..60u32 { + let s = format!("igneum-hot-accept/{i}"); + let q = candidate_class(&s, s.as_bytes(), 0, c); + if check(&q).is_err() { + rejected += 1; + } + } + assert!(rejected < 15, "{name}: {rejected} of 60 rejected"); + } + let p = generate_class("igneum-genesis", LoadClass::hot(96, 4)); + let words = p.hot_words(); + let loads = p.loads_per_hash(); + let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 }; + let mut la = vec![0u32; LANES * loads]; + let mut buckets = [0u64; 16]; + let mut hot_count = 0u64; + for (u, &b) in accept_base_nonces(&p.seed).iter().enumerate() { + run_unit(&p, u, b, &mut acc, &mut la).unwrap(); + for &a in &la { + if a & 0xC000_0000 == 0x4000_0000 { + let idx = a & 0x3FFF_FFFF; + assert!(idx < words); + buckets[(idx as u64 * 16 / words as u64) as usize] += 1; + hot_count += 1; + } + } + } + assert_eq!(hot_count, 64 * 32 * 32, "32 hot loads per hash over 2,048 hashes"); + let mean = hot_count as f64 / 16.0; + for (i, &b) in buckets.iter().enumerate() { + assert!((b as f64 - mean).abs() < 0.15 * mean, "bucket {i}: {b} against a mean of {mean}"); + } + // the dataset distinct count still holds for the dataset loads alone + let r = check(&p).unwrap(); + assert!(r.distinct_mean() > 120.0); + } + + #[test] + fn base_nonces_are_aligned_and_seed_dependent() { + let a = accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 8]); + let b = accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 9]); + assert!(a.iter().all(|x| x & 31 == 0)); + assert_ne!(a, b); + assert_eq!(a, accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 8])); + } + + #[test] + fn stale_load_detection_is_cyclic() { + let mut p = candidate("igneum-genesis", b"igneum-genesis", 0); + assert!(check_stale_loads(&p.instrs).is_ok(), "an accepted candidate has no stale load"); + // Make the last instruction a load from r3 and the first a load from r3 with no write between (wrap). + let (first, last) = (0usize, INSTR_COUNT - 1); + p.instrs[last].op = Op::Load; + p.instrs[last].src = 3; + p.instrs[last].dst = 4; + p.instrs[first].op = Op::Load; + p.instrs[first].src = 3; + p.instrs[first].dst = 5; + assert_eq!(check_stale_loads(&p.instrs), Err(Reject::StaleLoadSource { instr: 0, reg: 3 })); + } + + #[test] + fn injecting_write_detection() { + let mut p = candidate("igneum-genesis", b"igneum-genesis", 0); + for ins in p.instrs.iter_mut() { + if ins.dst == 6 && ins.op.injects() { + ins.op = Op::Rotl; + } + } + assert_eq!(check_injecting_writes(&p.instrs), Err(Reject::NoInjectingWrite { reg: 6 })); + } + + /// The census's measured rates: about 5 percent of candidates rejected, 128 distinct loads for the rest. + #[test] + fn rejection_rate_and_distinct_loads_on_a_sample() { + let mut rejected = 0; + let mut dsum = 0.0; + let mut accepted = 0; + for i in 0..400u32 { + let s = format!("igneum-census-2026-10-03/{i}"); + match check(&candidate(&s, s.as_bytes(), 0)) { + Ok(r) => { + accepted += 1; + dsum += r.distinct_mean(); + assert!(r.distinct_mean() > 120.0); + } + Err(_) => rejected += 1, + } + } + assert!(rejected < 50, "{rejected} of 400 rejected"); + assert!(dsum / accepted as f64 > 127.0, "mean distinct {}", dsum / accepted as f64); + } + + /// The retired generator fails the rule on nearly every program (the census: 95 percent). + #[test] + fn v1_programs_are_mostly_rejected() { + let mut rejected = 0; + for i in 0..100u32 { + let s = format!("igneum-census-2026-10-03/{i}"); + if check(&generate_v1(&s, &GeneratorConfig::default())).is_err() { + rejected += 1; + } + } + assert!(rejected > 80, "{rejected} of 100 rejected"); + } + + #[test] + fn generated_programs_pass() { + for s in ["igneum-genesis", "igneum-hourly", "igneum-second-seed"] { + assert!(check(&generate(s)).is_ok()); + } + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/bind.rs b/tools/attack/adv-accept-v5/igneum-pow/src/bind.rs new file mode 100644 index 000000000..94edea2bd --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/bind.rs @@ -0,0 +1,234 @@ +//! Header binding: the init words of the lottery hash commit to the block being mined. +//! +//! `docs/spec/01-lottery-hash.md` section 1.6 (open item O-1.9) proposes the rule implemented here: +//! +//! * The header nonce is 64 bits. Its low 32 bits are the lane nonce `n` (the per-thread nonce of the kernel). +//! * Its high 32 bits and the 256-bit pre-PoW header hash `H` form the init words: +//! `I = seed_words_from_bytes("igneum-block/" || H || nonce_hi_le32)`. +//! * `I` is a kernel argument, not a compile-time constant. The program (from the epoch seed) is compiled once per +//! epoch; `I` changes per block template. +//! +//! Choices the spec leaves open, fixed here (3 October 2026): +//! +//! | Choice | Rule | Why | +//! |---|---|---| +//! | `H` | the header hash with the nonce field set to zero and every other field as mined (rusty-kaspa `hash_override_nonce_time(header, 0, header.timestamp)`) | Kaspa's pre-PoW hash also zeroes the timestamp and absorbs it later in cSHAKE. The lane hash has no later step, so the timestamp must be inside `H` or a miner could reuse one nonce for many timestamps | +//! | 256-bit pow value | lane hash in the top 64 bits (little-endian bytes 24..32), low 192 bits zero | `pow <= target256` is then exactly `lane <= target256 >> 192` (section 1.10 candidate), so a GPU worker and the node compare the same 64-bit numbers; the block level is `leading_zeros(lane)` shifted | +//! | Day seed bytes (O-1.10 interim) | `"igneum-day/" || day_le64` with `day = header.timestamp_ms / 86,400,000` | The spec's proposal ties the day key to the first epoch seed of the day, which needs the VDF schedule. The interim rule keeps one 256 MiB cache per calendar day and needs no chain walk | +//! | Epoch seed bytes | the 32 bytes of the epoch block hash (devnet v0: the last selected-chain block below the epoch's start DAA score, genesis for epoch 0) | Section 1.12: `S_e = seed_words_from_bytes(program_seed_e)`; the VDF output replaces the block hash later without touching this crate | +//! +//! The packs' vectors (init words equal to the program seed) stay the conformance vectors for the generator, +//! interpreter and dataset. The bound vectors are in `README.md` and in the tests below (re-cut for generator +//! version 2 on 4 October 2026). + +use crate::generator::LANES; +use crate::seed::seed_words_from_bytes; +use crate::verify::{interpret_warp_init, Epoch, WarpResult}; + +/// Domain tag of the init words. +pub const BLOCK_TAG: &[u8] = b"igneum-block/"; +/// Domain tag of the interim day seed. +pub const DAY_TAG: &[u8] = b"igneum-day/"; +/// Milliseconds per day, the clock of the interim day seed. +pub const DAY_MS: u64 = 86_400_000; + +/// The lane nonce: low 32 bits of the header nonce. +#[inline] +pub fn lane_nonce(nonce: u64) -> u32 { + nonce as u32 +} + +/// The high 32 bits of the header nonce (the extra nonce that enters the init words). +#[inline] +pub fn nonce_hi(nonce: u64) -> u32 { + (nonce >> 32) as u32 +} + +/// `"igneum-block/" || H || nonce_hi_le32`, the bytes the init words are derived from. +pub fn block_init_bytes(header_prehash: &[u8; 32], nonce: u64) -> [u8; 49] { + let mut b = [0u8; 49]; + b[..13].copy_from_slice(BLOCK_TAG); + b[13..45].copy_from_slice(header_prehash); + b[45..49].copy_from_slice(&nonce_hi(nonce).to_le_bytes()); + b +} + +/// The init words `I` for a header and a 64-bit nonce (only the high 32 bits of the nonce matter). +pub fn block_init_words(header_prehash: &[u8; 32], nonce: u64) -> [u32; 8] { + seed_words_from_bytes(&block_init_bytes(header_prehash, nonce)) +} + +/// Interim day seed bytes: `"igneum-day/" || day_le64`. +pub fn day_bytes(day_index: u64) -> [u8; 19] { + let mut b = [0u8; 19]; + b[..11].copy_from_slice(DAY_TAG); + b[11..19].copy_from_slice(&day_index.to_le_bytes()); + b +} + +/// Day index of a header timestamp in milliseconds. +#[inline] +pub fn day_index(timestamp_ms: u64) -> u64 { + timestamp_ms / DAY_MS +} + +/// The 256-bit pow value as little-endian bytes: the lane hash in bytes 24..32, zero elsewhere. +pub fn pow256_from_lane(lane: u64) -> [u8; 32] { + let mut b = [0u8; 32]; + b[24..32].copy_from_slice(&lane.to_le_bytes()); + b +} + +/// The 64-bit target from a little-endian 256-bit target: its top 64 bits. +pub fn target64_from_le256(target: &[u8; 32]) -> u64 { + u64::from_le_bytes(target[24..32].try_into().unwrap()) +} + +/// Lower-case hex of bytes. +pub fn hex(bytes: &[u8]) -> String { + bytes.iter().map(|b| format!("{b:02x}")).collect() +} + +/// Bytes from hex (either case). `None` on odd length or a bad digit. +pub fn unhex(s: &str) -> Option> { + if s.len() % 2 != 0 { + return None; + } + (0..s.len()).step_by(2).map(|i| u8::from_str_radix(&s[i..i + 2], 16).ok()).collect() +} + +impl Epoch { + /// The 32 bound hashes of the aligned warp that contains `nonce`: lane `l` is the hash of + /// `(nonce_hi << 32) | ((lane_nonce & !31) + l)`. + pub fn hash_warp_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> [u64; LANES] { + self.interpret_warp_bound(header_prehash, nonce).hashes + } + + pub fn interpret_warp_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> WarpResult { + let init = block_init_words(header_prehash, nonce); + interpret_warp_init(&self.program, &init, lane_nonce(nonce) & !31, &self.dataset) + } + + /// The 32 bound hashes for already-derived init words (what a GPU worker computes per dispatch). + pub fn hash_warp_init(&self, init: &[u32; 8], base_lane_nonce: u32) -> [u64; LANES] { + interpret_warp_init(&self.program, init, base_lane_nonce, &self.dataset).hashes + } + + /// The bound 64-bit lane hash of one header nonce. + pub fn hash_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> u64 { + self.hash_warp_bound(header_prehash, nonce)[(lane_nonce(nonce) & 31) as usize] + } + + /// The bound 256-bit pow value (little-endian): the lane hash in the top 64 bits. + pub fn pow_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> [u8; 32] { + pow256_from_lane(self.hash_bound(header_prehash, nonce)) + } + + /// `hash_bound(H, nonce) <= target64`. + pub fn verify_block_bound(&self, header_prehash: &[u8; 32], nonce: u64, target64: u64) -> bool { + self.hash_bound(header_prehash, nonce) <= target64 + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::verify::DatasetMode; + use std::sync::OnceLock; + + fn epoch() -> &'static Epoch { + static E: OnceLock = OnceLock::new(); + E.get_or_init(|| Epoch::memory_hard("igneum-genesis", "2026-10-03")) + } + + fn prehash_a() -> [u8; 32] { + [0u8; 32] + } + fn prehash_b() -> [u8; 32] { + let mut h = [0u8; 32]; + for (i, b) in h.iter_mut().enumerate() { + *b = i as u8; + } + h + } + + #[test] + fn init_bytes_layout() { + let b = block_init_bytes(&prehash_b(), 0x0000_0102_0000_0007); + assert_eq!(&b[..13], b"igneum-block/"); + assert_eq!(&b[13..45], &prehash_b()); + assert_eq!(&b[45..49], &[0x02, 0x01, 0x00, 0x00]); + assert_eq!( + block_init_words(&prehash_b(), 0x0000_0102_0000_0007), + block_init_words(&prehash_b(), 0x0000_0102_ffff_ffff) + ); + assert_ne!(block_init_words(&prehash_b(), 0), block_init_words(&prehash_b(), 1 << 32)); + } + + #[test] + fn day_bytes_layout() { + let b = day_bytes(20_729); + assert_eq!(&b[..11], b"igneum-day/"); + assert_eq!(&b[11..], &20_729u64.to_le_bytes()); + assert_eq!(day_index(0x1a0ff0f7c00), 20_729); + } + + #[test] + fn pow256_and_target64() { + let p = pow256_from_lane(0x0123_4567_89ab_cdef); + assert_eq!(&p[..24], &[0u8; 24]); + assert_eq!(target64_from_le256(&p), 0x0123_4567_89ab_cdef); + assert_eq!(hex(&p[24..]), "efcdab8967452301"); + assert_eq!(unhex("efcdab8967452301").unwrap(), p[24..].to_vec()); + assert!(unhex("abc").is_none()); + } + + /// The bound form is a different hash from the unbound one, depends on H and on the high nonce bits, and the + /// single-nonce form agrees with the warp form. + #[test] + fn bound_hash_properties() { + let e = epoch(); + let a = prehash_a(); + let b = prehash_b(); + assert_ne!(e.hash_bound(&a, 0), e.hash(0), "bound differs from the pack vector"); + assert_ne!(e.hash_bound(&a, 0), e.hash_bound(&b, 0), "H enters the hash"); + assert_ne!(e.hash_bound(&a, 0), e.hash_bound(&a, 1 << 32), "nonce_hi enters the hash"); + let w = e.hash_warp_bound(&a, (1 << 32) | 37); + assert_eq!(w[5], e.hash_bound(&a, (1 << 32) | 37)); + assert_eq!(w[0], e.hash_bound(&a, 1 << 32 | 32)); + let init = block_init_words(&a, 1 << 32); + assert_eq!(e.hash_warp_init(&init, 32), w); + let t = e.hash_bound(&a, 7); + assert!(e.verify_block_bound(&a, 7, t)); + assert!(!e.verify_block_bound(&a, 7, t - 1)); + assert_eq!(e.pow_bound(&a, 7), pow256_from_lane(t)); + } + + /// The 8 bound vectors printed in README.md (seed igneum-genesis, day 2026-10-03, memory-hard, 2^28 words, + /// generator v2 since 4 October 2026). + #[test] + fn bound_vectors() { + let e = epoch(); + let a = prehash_a(); + let b = prehash_b(); + let cases: [(&[u8; 32], u64, u64); 8] = [ + (&a, 0, 0x746c567b090acf6a), + (&a, 1, 0x45a619f860880c73), + (&a, 31, 0x9aa495e43dedbfe6), + (&a, 4096, 0x2e6ffd7624d3cba2), + (&a, 1 << 32, 0x38a5cea1fb01431a), + (&b, 0, 0x2a79c5e4797bf6aa), + (&b, (1 << 32) | 5, 0xa243e0c61aa1b82e), + (&b, u64::MAX, 0x9c7bbfbd064fe1a4), + ]; + for (h, nonce, want) in cases { + assert_eq!(e.hash_bound(h, nonce), want, "H {} nonce {nonce}", hex(h)); + } + } + + #[test] + fn closed_form_bound_also_works() { + let e = Epoch::new("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 28); + assert_ne!(e.hash_bound(&prehash_a(), 0), e.hash(0)); + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/blake2b.rs b/tools/attack/adv-accept-v5/igneum-pow/src/blake2b.rs new file mode 100644 index 000000000..4b04cc622 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/blake2b.rs @@ -0,0 +1,152 @@ +//! BLAKE2b (RFC 7693), the chain's own hash family (spec 01 section 0.6), written out here so the crate keeps its +//! rule of no dependency outside the standard library. Used by class v5's state leaves (`crate::state`): +//! `blake2b_512` for a leaf digest, `blake2b_256` for the sample order. Unkeyed, no salt, no personalisation. +//! Checked against the RFC's "abc" vector and the empty-input vector in the tests. + +const IV: [u64; 8] = [ + 0x6a09e667f3bcc908, + 0xbb67ae8584caa73b, + 0x3c6ef372fe94f82b, + 0xa54ff53a5f1d36f1, + 0x510e527fade682d1, + 0x9b05688c2b3e6c1f, + 0x1f83d9abfb41bd6b, + 0x5be0cd19137e2179, +]; + +const SIGMA: [[usize; 16]; 12] = [ + [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15], + [14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3], + [11, 8, 12, 0, 5, 2, 15, 13, 10, 14, 3, 6, 7, 1, 9, 4], + [7, 9, 3, 1, 13, 12, 11, 14, 2, 6, 5, 10, 4, 0, 15, 8], + [9, 0, 5, 7, 2, 4, 10, 15, 14, 1, 11, 12, 6, 8, 3, 13], + [2, 12, 6, 10, 0, 11, 8, 3, 4, 13, 7, 5, 15, 14, 1, 9], + [12, 5, 1, 15, 14, 13, 4, 10, 0, 7, 6, 3, 9, 2, 8, 11], + [13, 11, 7, 14, 12, 1, 3, 9, 5, 0, 15, 4, 8, 6, 2, 10], + [6, 15, 14, 9, 11, 3, 0, 8, 12, 2, 13, 7, 1, 4, 10, 5], + [10, 2, 8, 4, 7, 6, 1, 5, 15, 11, 9, 14, 3, 12, 13, 0], + [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15], + [14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3], +]; + +#[inline(always)] +fn g(v: &mut [u64; 16], a: usize, b: usize, c: usize, d: usize, x: u64, y: u64) { + v[a] = v[a].wrapping_add(v[b]).wrapping_add(x); + v[d] = (v[d] ^ v[a]).rotate_right(32); + v[c] = v[c].wrapping_add(v[d]); + v[b] = (v[b] ^ v[c]).rotate_right(24); + v[a] = v[a].wrapping_add(v[b]).wrapping_add(y); + v[d] = (v[d] ^ v[a]).rotate_right(16); + v[c] = v[c].wrapping_add(v[d]); + v[b] = (v[b] ^ v[c]).rotate_right(63); +} + +fn compress(h: &mut [u64; 8], block: &[u8; 128], t: u128, last: bool) { + let mut m = [0u64; 16]; + for (i, w) in m.iter_mut().enumerate() { + *w = u64::from_le_bytes(block[i * 8..i * 8 + 8].try_into().unwrap()); + } + let mut v = [0u64; 16]; + v[..8].copy_from_slice(h); + v[8..].copy_from_slice(&IV); + v[12] ^= t as u64; + v[13] ^= (t >> 64) as u64; + if last { + v[14] = !v[14]; + } + for s in SIGMA.iter() { + g(&mut v, 0, 4, 8, 12, m[s[0]], m[s[1]]); + g(&mut v, 1, 5, 9, 13, m[s[2]], m[s[3]]); + g(&mut v, 2, 6, 10, 14, m[s[4]], m[s[5]]); + g(&mut v, 3, 7, 11, 15, m[s[6]], m[s[7]]); + g(&mut v, 0, 5, 10, 15, m[s[8]], m[s[9]]); + g(&mut v, 1, 6, 11, 12, m[s[10]], m[s[11]]); + g(&mut v, 2, 7, 8, 13, m[s[12]], m[s[13]]); + g(&mut v, 3, 4, 9, 14, m[s[14]], m[s[15]]); + } + for i in 0..8 { + h[i] ^= v[i] ^ v[i + 8]; + } +} + +/// Unkeyed BLAKE2b of `data` with an output of `out_len` bytes (1..=64), written into `out[..out_len]`. +pub fn blake2b(out: &mut [u8], out_len: usize, data: &[u8]) { + assert!((1..=64).contains(&out_len) && out.len() >= out_len); + let mut h = IV; + h[0] ^= 0x0101_0000 ^ out_len as u64; + let mut t: u128 = 0; + let n = data.len(); + // every full block but the last; the last block (possibly empty) is compressed with the final flag + let full = if n == 0 { 0 } else { (n - 1) / 128 }; + for i in 0..full { + let block: &[u8; 128] = data[i * 128..i * 128 + 128].try_into().unwrap(); + t += 128; + compress(&mut h, block, t, false); + } + let mut last = [0u8; 128]; + let rest = &data[full * 128..]; + last[..rest.len()].copy_from_slice(rest); + t += rest.len() as u128; + compress(&mut h, &last, t, true); + let mut bytes = [0u8; 64]; + for (i, w) in h.iter().enumerate() { + bytes[i * 8..i * 8 + 8].copy_from_slice(&w.to_le_bytes()); + } + out[..out_len].copy_from_slice(&bytes[..out_len]); +} + +/// BLAKE2b-512 of the concatenation of `parts`. +pub fn blake2b_512(parts: &[&[u8]]) -> [u8; 64] { + let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum()); + for p in parts { + data.extend_from_slice(p); + } + let mut out = [0u8; 64]; + blake2b(&mut out, 64, &data); + out +} + +/// BLAKE2b-256 of the concatenation of `parts`. +pub fn blake2b_256(parts: &[&[u8]]) -> [u8; 32] { + let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum()); + for p in parts { + data.extend_from_slice(p); + } + let mut out = [0u8; 32]; + blake2b(&mut out, 32, &data); + out +} + +#[cfg(test)] +mod tests { + use super::*; + + fn hex(b: &[u8]) -> String { + b.iter().map(|x| format!("{x:02x}")).collect() + } + + /// RFC 7693 appendix A ("abc"), the empty input, and a two-block input against the reference implementation's + /// known values (the three-block "The quick brown fox" vector of the BLAKE2 test suite). + #[test] + fn rfc_7693_vectors() { + assert_eq!( + hex(&blake2b_512(&[b"abc"])), + "ba80a53f981c4d0d6a2797b69f12f6e94c212f14685ac4b74b12bb6fdbffa2d17d87c5392aab792dc252d5de4533cc9518d38aa8dbf1925ab92386edd4009923" + ); + assert_eq!( + hex(&blake2b_512(&[b""])), + "786a02f742015903c6c6fd852552d272912f4740e15847618a86e217f71f5419d25e1031afee585313896444934eb04b903a685b1448b755d56f701afe9be2ce" + ); + assert_eq!(hex(&blake2b_256(&[b"abc"])), "bddd813c634239723171ef3fee98579b94964e3bb1cb3e427262c8c068d52319"); + assert_eq!(hex(&blake2b_256(&[b""])), "0e5751c026e543b2e8ab2eb06099daa1d1e5df47778f7787faab45cdf12fe3a8"); + // a 128-byte input is exactly one full block compressed as the last; 129 bytes takes two + let one = [0x61u8; 128]; + let two = [0x61u8; 129]; + assert_ne!(blake2b_512(&[&one]), blake2b_512(&[&two])); + assert_eq!(blake2b_512(&[&one[..64], &one[64..]]), blake2b_512(&[&one]), "parts concatenate"); + assert_eq!( + hex(&blake2b_512(&[b"The quick brown fox jumps over the lazy dog"])), + "a8add4bdddfd93e4877d2746e62817b116364a1fa7bc148d95090bc7333b3673f82401cf7aa2e4cb1ecd90296e3f14cb5413f8ed77be73045b13914cdcd6a918" + ); + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/derive.rs b/tools/attack/adv-accept-v5/igneum-pow/src/derive.rs new file mode 100644 index 000000000..7e8c5021f --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/derive.rs @@ -0,0 +1,978 @@ +//! The per-day item-derivation program (Counter ASIC 3.0 item 2, `docs/plans/counter-asic-3-derivation.md`): +//! RandomX's SuperscalarHash idea (`vendor/RandomX/src/superscalar.cpp`, read 6 October 2026 at commit 7607fb2) +//! rebuilt for a 16-word item on a GPU. In place of the fixed-shape mixer `M_r` of spec 01 section 1.8.4, each of +//! the nine mixer slots of an item (one before each of the 8 dependent cache reads, one after the last) runs a +//! straight-line program of [`DERIVE_LEN`] instructions drawn once a day from the day key stream, from a fixed +//! set of twelve two-register forms. The 8 dependent cache reads per item are untouched. +//! +//! Rules of the draw (the dependency chain of SuperscalarHash, made strict): +//! * every instruction reads the chain register `c`, the register the previous instruction wrote (`s[0]`, the +//! address word, at the start of each round program), and writes a register `d != c`, which becomes the chain; +//! so no two instructions of a program can run in parallel, and no two consecutive instructions write one +//! register (the "ror r,C1; ror r,C2" and "xor r,r2; xor r,r2" merges of SuperscalarHash's `selectDestination` +//! cannot arise); +//! * every form is a bijection on the 16-word state (the old `d` enters through `+=`, `-=`, `^=`, an odd multiply, +//! or a rotation of itself), so a program loses no entropy, the property `M_r` has; +//! * the forms are integer only, modulo 2^32, with rotations by 1..31: bit-exact on Metal, CUDA and OpenCL by the +//! same argument as the lottery hash's families (spec 01 section 1.14); no division, no float, no branch; +//! * four draws per instruction in a fixed order, so the stream position of every draw is fixed by the index. +//! +//! The acceptance test ([`DeriveProgram::check`]) rejects a degenerate draw and the next attempt is drawn from the +//! continuation of the stream, the rule the program generator uses (spec 01 section 1.4.6). + +use crate::memhard::ITEM_ROUNDS; +use crate::seed::SplitMix64; + +/// Registers of the item state (the item is 16 words). +pub const DERIVE_REGS: usize = 16; +/// Round programs per item: one before each cache read and one after the last (`ITEM_ROUNDS + 1`). +pub const DERIVE_PROGRAMS: usize = ITEM_ROUNDS + 1; +/// Instructions per round program for the x8-equivalent operation count (the candidate, class "dr736"): 9 x 736 +/// = 6,624 instructions per item at a mean of 1.62 GPU operations (1.52 chip operations) each, about 10,730 GPU +/// operations, 10,070 chip operations and 1,460 multiplies per item. The x8 mixer, counted from the code +/// (`memhard::mixer`, 16 x (xor, add, mul) + 8 quarter rounds x 12 = 144 operations as written, 128 with the +/// `RC + rk` adds hoisted as constants, 16 multiplies; `chip-model-v3.md` section 1 prices 130 from the spec text): +/// 72 applications = 10,368 as written, 9,216 hoisted, 1,152 multiplies. The floors below are those three. +pub const DERIVE_LEN_X8: u32 = 736; +/// Draws per instruction: the op roll, the destination roll, the second-source roll and the immediate. +pub const DRAWS_PER_INSTR: u64 = 4; +/// The floor of chip operations per item (the x8 mixer with its constants hoisted: 72 x 128). +pub const OPS_FLOOR_X8: u64 = 9_216; +/// The floor of GPU operations per item (the x8 mixer as written: 72 x 144). +pub const GPU_OPS_FLOOR_X8: u64 = 10_368; +/// The floor of multiplies per item (the x8 mixer's 72 x 16). +pub const MULS_FLOOR_X8: u64 = 1_152; +/// Distinct rotation amounts an item's programs must use, at least. +pub const DISTINCT_ROTS_FLOOR: usize = 8; +/// Attempts before the generator gives up (never reached: see [`DeriveProgram::draw`]). +pub const MAX_ATTEMPTS: u32 = 64; + +/// The twelve forms. `c` is the chain register (the previous destination), `d` the destination (`d != c`), `b` a +/// third register (`b != d`, `b != c`), `k` a rotation in 1..31, `i` a 32-bit constant (odd for `MulC`). +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +#[repr(u8)] +pub enum DOp { + /// `d += c` + Add = 0, + /// `d -= c` + Sub = 1, + /// `d ^= c` + Xor = 2, + /// `d *= (c OR 1)`: the multiply-lo form, odd so it is a bijection on `d` + Mul = 3, + /// `d = rotl(d, k) + c` + Rot = 4, + /// `d = rotl(d ^ c, k)` + XRot = 5, + /// `d += c + i` + AddC = 6, + /// `d ^= c ^ i` + XorC = 7, + /// `d = (d ^ c) * i`, `i` odd: the per-word form of `M_r` with the chain in place of the round constant + MulC = 8, + /// `d = d * i + c`, `i` odd + MulC2 = 9, + /// `d ^= (c AND b)` + AndX = 10, + /// `d += (c OR b)` + OrX = 11, +} + +/// The op weights in percent, in draw order (sum 100). Fixed at genesis; only the order, the registers and the +/// constants are drawn. +pub const DOP_WEIGHTS: [(DOp, u64); 12] = [ + (DOp::Add, 14), + (DOp::Sub, 10), + (DOp::Xor, 14), + (DOp::Mul, 10), + (DOp::Rot, 10), + (DOp::XRot, 10), + (DOp::AddC, 6), + (DOp::XorC, 6), + (DOp::MulC, 8), + (DOp::MulC2, 4), + (DOp::AndX, 4), + (DOp::OrX, 4), +]; + +impl DOp { + pub fn from_u8(v: u8) -> Option { + DOP_WEIGHTS.iter().map(|(o, _)| *o).find(|o| *o as u8 == v) + } + pub fn name(self) -> &'static str { + match self { + DOp::Add => "add", + DOp::Sub => "sub", + DOp::Xor => "xor", + DOp::Mul => "mul", + DOp::Rot => "rot", + DOp::XRot => "xrot", + DOp::AddC => "addc", + DOp::XorC => "xorc", + DOp::MulC => "mulc", + DOp::MulC2 => "mulc2", + DOp::AndX => "andx", + DOp::OrX => "orx", + } + } + /// Integer operations as a GPU executes the form (every `|`, `&`, `+`, `^`, `*`, rotate counts one). + pub fn gpu_ops(self) -> u64 { + match self { + DOp::Add | DOp::Sub | DOp::Xor => 1, + _ => 2, + } + } + /// Integer operations as the chip model counts them (`c OR 1` is a wire on a chip, so `Mul` is one multiply; + /// a constant folded into a chain value is still an add or an xor, so every other two-op form stays two). + pub fn chip_ops(self) -> u64 { + match self { + DOp::Add | DOp::Sub | DOp::Xor | DOp::Mul => 1, + _ => 2, + } + } + pub fn is_mul(self) -> bool { + matches!(self, DOp::Mul | DOp::MulC | DOp::MulC2) + } + pub fn has_rot(self) -> bool { + matches!(self, DOp::Rot | DOp::XRot) + } + pub fn has_third(self) -> bool { + matches!(self, DOp::AndX | DOp::OrX) + } + pub fn has_imm(self) -> bool { + matches!(self, DOp::AddC | DOp::XorC | DOp::MulC | DOp::MulC2) + } + /// The op of a roll in 0..99. + pub fn for_roll(roll: u64) -> DOp { + let mut acc = 0u64; + for (op, w) in DOP_WEIGHTS { + acc += w; + if roll < acc { + return op; + } + } + DOp::OrX + } +} + +/// One instruction. `src` is the chain register (carried so the interpreter and the emitter need no state). +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct DInstr { + pub op: DOp, + pub dst: u8, + pub src: u8, + /// The third register of `AndX` and `OrX`; 0 on every other form (drawn and unused). + pub src2: u8, + /// The rotation 1..31 of `Rot` and `XRot`, else 0. + pub rot: u8, + /// The constant of `AddC`, `XorC` (any), `MulC` and `MulC2` (odd); 0 on every other form. + pub imm: u32, +} + +/// The nine round programs of an item for one day, with the attempt that passed the acceptance test. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct DeriveProgram { + pub len: u32, + pub attempt: u32, + pub rounds: Vec>, +} + +/// Why a candidate was rejected. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum DeriveReject { + /// A register no instruction of round program `round` writes. + RegisterNeverWritten { round: usize, reg: u8 }, + /// Fewer than [`DISTINCT_ROTS_FLOOR`] distinct rotation amounts over the item's programs. + RotationsDegenerate { distinct: usize }, + /// Chip operations per item under [`OPS_FLOOR_X8`] scaled to the length. + OpsUnderFloor { ops: u64, floor: u64 }, + /// GPU operations per item under [`GPU_OPS_FLOOR_X8`] scaled to the length. + GpuOpsUnderFloor { ops: u64, floor: u64 }, + /// Multiplies per item under [`MULS_FLOOR_X8`] scaled to the length. + MulsUnderFloor { muls: u64, floor: u64 }, +} + +impl std::fmt::Display for DeriveReject { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + DeriveReject::RegisterNeverWritten { round, reg } => write!(f, "register {reg} never written in round program {round}"), + DeriveReject::RotationsDegenerate { distinct } => write!(f, "only {distinct} distinct rotation amounts"), + DeriveReject::OpsUnderFloor { ops, floor } => write!(f, "{ops} chip operations per item, floor {floor}"), + DeriveReject::GpuOpsUnderFloor { ops, floor } => write!(f, "{ops} GPU operations per item, floor {floor}"), + DeriveReject::MulsUnderFloor { muls, floor } => write!(f, "{muls} multiplies per item, floor {floor}"), + } + } +} + +impl DeriveProgram { + /// Draw one candidate of `len` instructions per round program from `rng` (four draws per instruction). + pub fn draw_candidate(rng: &mut SplitMix64, len: u32, attempt: u32) -> DeriveProgram { + let mut rounds = Vec::with_capacity(DERIVE_PROGRAMS); + for _ in 0..DERIVE_PROGRAMS { + let mut prog = Vec::with_capacity(len as usize); + let mut chain = 0u8; + for _ in 0..len { + let op = DOp::for_roll(rng.below(100)); + // the destination: the 15 registers other than the chain, in ascending order + let d_roll = rng.below((DERIVE_REGS - 1) as u64) as u8; + let dst = if d_roll >= chain { d_roll + 1 } else { d_roll }; + // the third register: the 14 registers other than dst and the chain, in ascending order + let b_roll = rng.below((DERIVE_REGS - 2) as u64) as u8; + let (lo, hi) = if dst < chain { (dst, chain) } else { (chain, dst) }; + let mut b = b_roll; + if b >= lo { + b += 1; + } + if b >= hi { + b += 1; + } + let x = rng.next() as u32; + let mut ins = DInstr { op, dst, src: chain, src2: 0, rot: 0, imm: 0 }; + if op.has_third() { + ins.src2 = b; + } + if op.has_rot() { + ins.rot = 1 + (x % 31) as u8; + } + if op.has_imm() { + ins.imm = if matches!(op, DOp::MulC | DOp::MulC2) { x | 1 } else { x }; + } + prog.push(ins); + chain = dst; + } + rounds.push(prog); + } + DeriveProgram { len, attempt, rounds } + } + + /// Draw the program of a day: candidates from `rng` in turn until one passes [`DeriveProgram::check`]. + /// Panics after [`MAX_ATTEMPTS`] (the floors sit more than 7 standard deviations under the expected counts, so + /// a rejection is a rare event and 64 in a row is not one that happens). + pub fn draw(rng: &mut SplitMix64, len: u32) -> DeriveProgram { + for attempt in 0..MAX_ATTEMPTS { + let p = Self::draw_candidate(rng, len, attempt); + if p.check().is_ok() { + return p; + } + } + panic!("derivation program: {MAX_ATTEMPTS} candidates rejected in a row"); + } + + /// The floors for this length (chip operations, GPU operations, multiplies): the x8 floors scaled by + /// `len / DERIVE_LEN_X8`, so a shorter class, measured as a fallback, has its own proportional floors. + pub fn floors(len: u32) -> (u64, u64, u64) { + let scale = |f: u64| f * len as u64 / DERIVE_LEN_X8 as u64; + (scale(OPS_FLOOR_X8), scale(GPU_OPS_FLOOR_X8), scale(MULS_FLOOR_X8)) + } + + /// The acceptance test: every register written in every round program; at least [`DISTINCT_ROTS_FLOOR`] + /// distinct rotation amounts; chip operations, GPU operations and multiplies per item at or above the floors + /// (the x8 mixer's counts from the code). The structural + /// rules (`dst != src`, the third register distinct, rotations in 1..31, odd multiplier constants) hold by + /// construction and are asserted. + pub fn check(&self) -> Result<(), DeriveReject> { + let mut rots = [false; 32]; + for (r, prog) in self.rounds.iter().enumerate() { + let mut written = [false; DERIVE_REGS]; + let mut chain = 0u8; + for ins in prog { + assert!(ins.src == chain && ins.dst != ins.src && (ins.dst as usize) < DERIVE_REGS, "chain rule"); + if ins.op.has_third() { + assert!(ins.src2 != ins.dst && ins.src2 != ins.src && (ins.src2 as usize) < DERIVE_REGS, "third register"); + } + if ins.op.has_rot() { + assert!((1..=31).contains(&ins.rot), "rotation"); + rots[ins.rot as usize] = true; + } + if matches!(ins.op, DOp::MulC | DOp::MulC2) { + assert!(ins.imm & 1 == 1, "odd multiplier"); + } + written[ins.dst as usize] = true; + chain = ins.dst; + } + if let Some(reg) = written.iter().position(|w| !w) { + return Err(DeriveReject::RegisterNeverWritten { round: r, reg: reg as u8 }); + } + } + let distinct = rots.iter().filter(|r| **r).count(); + if distinct < DISTINCT_ROTS_FLOOR { + return Err(DeriveReject::RotationsDegenerate { distinct }); + } + let (ops_floor, gpu_floor, muls_floor) = Self::floors(self.len); + let ops = self.chip_ops(); + if ops < ops_floor { + return Err(DeriveReject::OpsUnderFloor { ops, floor: ops_floor }); + } + let gpu = self.gpu_ops(); + if gpu < gpu_floor { + return Err(DeriveReject::GpuOpsUnderFloor { ops: gpu, floor: gpu_floor }); + } + let muls = self.muls(); + if muls < muls_floor { + return Err(DeriveReject::MulsUnderFloor { muls, floor: muls_floor }); + } + Ok(()) + } + + pub fn instr_count(&self) -> u64 { + self.rounds.iter().map(|p| p.len() as u64).sum() + } + pub fn gpu_ops(&self) -> u64 { + self.rounds.iter().flatten().map(|i| i.op.gpu_ops()).sum() + } + pub fn chip_ops(&self) -> u64 { + self.rounds.iter().flatten().map(|i| i.op.chip_ops()).sum() + } + pub fn muls(&self) -> u64 { + self.rounds.iter().flatten().filter(|i| i.op.is_mul()).count() as u64 + } + /// Count per op, in [`DOP_WEIGHTS`] order. + pub fn op_counts(&self) -> [u64; 12] { + let mut c = [0u64; 12]; + for i in self.rounds.iter().flatten() { + c[i.op as usize] += 1; + } + c + } + /// "add=887 sub=..." in weight order. + pub fn op_mix(&self) -> String { + let c = self.op_counts(); + DOP_WEIGHTS.iter().map(|(o, _)| format!("{}={}", o.name(), c[*o as usize])).collect::>().join(" ") + } + /// FNV-1a 64 over the instruction stream (op, dst, src, src2, rot, imm as bytes): the program's fingerprint + /// for packs and logs. + pub fn fingerprint(&self) -> u64 { + let mut b = Vec::with_capacity(self.instr_count() as usize * 9); + for i in self.rounds.iter().flatten() { + b.push(i.op as u8); + b.push(i.dst); + b.push(i.src); + b.push(i.src2); + b.push(i.rot); + b.extend_from_slice(&i.imm.to_le_bytes()); + } + crate::seed::fnv1a64(&b) + } +} + +/// Lanes of the SoA interpreter: the verifier derives up to 32 distinct items per load (one per lane of the +/// unit), so each instruction runs across 32 item states at once and the dispatch is paid once per 32 items. +pub const SOA_LANES: usize = 32; + +/// The item states of a batch, word-major: `st[reg][lane]`. +pub type SoaState = [[u32; SOA_LANES]; DERIVE_REGS]; + +#[inline(always)] +fn rotl(x: u32, n: u32) -> u32 { + x.rotate_left(n) +} + +/// The twelve forms over a batch, one function each, every one a straight loop over the lanes the compiler +/// vectorises. The destination row and the source rows are distinct by the chain rule (`dst != src`, and the third +/// register distinct from both: asserted by [`DeriveProgram::check`] and checked here in debug builds), so the +/// rows are addressed through raw pointers rather than copied out of the state. +mod forms { + use super::{rotl, DInstr, SoaState, SOA_LANES}; + #[inline(always)] + pub fn add(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src); + + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + + *d = d.wrapping_add(*c); + } + } + } + #[inline(always)] + pub fn sub(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src); + + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + + *d = d.wrapping_sub(*c); + } + } + } + #[inline(always)] + pub fn xor(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src); + + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + + *d ^= *c; + } + } + } + #[inline(always)] + pub fn mul(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src); + + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + + *d = d.wrapping_mul(*c | 1); + } + } + } + #[inline(always)] + pub fn rot(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src); + let r = ins.rot as u32; + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + + *d = rotl(*d, r).wrapping_add(*c); + } + } + } + #[inline(always)] + pub fn xrot(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src); + let r = ins.rot as u32; + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + + *d = rotl(*d ^ *c, r); + } + } + } + #[inline(always)] + pub fn addc(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src); + let i = ins.imm; + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + + *d = d.wrapping_add(c.wrapping_add(i)); + } + } + } + #[inline(always)] + pub fn xorc(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src); + let i = ins.imm; + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + + *d ^= *c ^ i; + } + } + } + #[inline(always)] + pub fn mulc(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src); + let i = ins.imm; + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + + *d = (*d ^ *c).wrapping_mul(i); + } + } + } + #[inline(always)] + pub fn mulc2(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src); + let i = ins.imm; + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + + *d = d.wrapping_mul(i).wrapping_add(*c); + } + } + } + #[inline(always)] + pub fn andx(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src && ins.src2 != ins.dst && ins.src2 != ins.src); + + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + let bp = st.as_ptr().add(ins.src2 as usize) as *const u32; + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + let b = &*bp.add(k); + *d ^= *c & *b; + } + } + } + #[inline(always)] + pub fn orx(ins: &DInstr, st: &mut SoaState) { + debug_assert!(ins.dst != ins.src && ins.src2 != ins.dst && ins.src2 != ins.src); + + // SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the + // pointers stay inside `st`. + unsafe { + let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32; + let cp = st.as_ptr().add(ins.src as usize) as *const u32; + let bp = st.as_ptr().add(ins.src2 as usize) as *const u32; + for k in 0..SOA_LANES { + let d = &mut *dp.add(k); + let c = &*cp.add(k); + let b = &*bp.add(k); + *d = d.wrapping_add(*c | *b); + } + } + } +} + +/// One instruction over the batch (the single dispatch; [`run_round`] dispatches on pairs). +#[inline(always)] +pub fn run_instr(ins: &DInstr, st: &mut SoaState) { + match ins.op { + DOp::Add => forms::add(ins, st), + DOp::Sub => forms::sub(ins, st), + DOp::Xor => forms::xor(ins, st), + DOp::Mul => forms::mul(ins, st), + DOp::Rot => forms::rot(ins, st), + DOp::XRot => forms::xrot(ins, st), + DOp::AddC => forms::addc(ins, st), + DOp::XorC => forms::xorc(ins, st), + DOp::MulC => forms::mulc(ins, st), + DOp::MulC2 => forms::mulc2(ins, st), + DOp::AndX => forms::andx(ins, st), + DOp::OrX => forms::orx(ins, st), + } +} + +/// Run one round program over the batch. The dispatch is on PAIRS of instructions (144 arms, one indirect branch +/// per two instructions): the op sequence of a drawn program is random, so the branch predictor misses most +/// dispatches, and the miss (about 3.7 ns of the 7.2 ns an instruction cost per batch on one M5 Max core, measured +/// with `examples/derive_perf.rs` on 6 October 2026, a functional run) is paid once per pair instead of once per +/// instruction. The result is bit for bit that of [`run_instr`] in sequence. +#[inline(never)] +pub fn run_round(prog: &[DInstr], st: &mut SoaState) { + let mut it = prog.chunks_exact(2); + for pair in &mut it { + let (a, b) = (&pair[0], &pair[1]); + match (a.op as u8) * 12 + b.op as u8 { + 0 => { forms::add(a, st); forms::add(b, st); } + 1 => { forms::add(a, st); forms::sub(b, st); } + 2 => { forms::add(a, st); forms::xor(b, st); } + 3 => { forms::add(a, st); forms::mul(b, st); } + 4 => { forms::add(a, st); forms::rot(b, st); } + 5 => { forms::add(a, st); forms::xrot(b, st); } + 6 => { forms::add(a, st); forms::addc(b, st); } + 7 => { forms::add(a, st); forms::xorc(b, st); } + 8 => { forms::add(a, st); forms::mulc(b, st); } + 9 => { forms::add(a, st); forms::mulc2(b, st); } + 10 => { forms::add(a, st); forms::andx(b, st); } + 11 => { forms::add(a, st); forms::orx(b, st); } + 12 => { forms::sub(a, st); forms::add(b, st); } + 13 => { forms::sub(a, st); forms::sub(b, st); } + 14 => { forms::sub(a, st); forms::xor(b, st); } + 15 => { forms::sub(a, st); forms::mul(b, st); } + 16 => { forms::sub(a, st); forms::rot(b, st); } + 17 => { forms::sub(a, st); forms::xrot(b, st); } + 18 => { forms::sub(a, st); forms::addc(b, st); } + 19 => { forms::sub(a, st); forms::xorc(b, st); } + 20 => { forms::sub(a, st); forms::mulc(b, st); } + 21 => { forms::sub(a, st); forms::mulc2(b, st); } + 22 => { forms::sub(a, st); forms::andx(b, st); } + 23 => { forms::sub(a, st); forms::orx(b, st); } + 24 => { forms::xor(a, st); forms::add(b, st); } + 25 => { forms::xor(a, st); forms::sub(b, st); } + 26 => { forms::xor(a, st); forms::xor(b, st); } + 27 => { forms::xor(a, st); forms::mul(b, st); } + 28 => { forms::xor(a, st); forms::rot(b, st); } + 29 => { forms::xor(a, st); forms::xrot(b, st); } + 30 => { forms::xor(a, st); forms::addc(b, st); } + 31 => { forms::xor(a, st); forms::xorc(b, st); } + 32 => { forms::xor(a, st); forms::mulc(b, st); } + 33 => { forms::xor(a, st); forms::mulc2(b, st); } + 34 => { forms::xor(a, st); forms::andx(b, st); } + 35 => { forms::xor(a, st); forms::orx(b, st); } + 36 => { forms::mul(a, st); forms::add(b, st); } + 37 => { forms::mul(a, st); forms::sub(b, st); } + 38 => { forms::mul(a, st); forms::xor(b, st); } + 39 => { forms::mul(a, st); forms::mul(b, st); } + 40 => { forms::mul(a, st); forms::rot(b, st); } + 41 => { forms::mul(a, st); forms::xrot(b, st); } + 42 => { forms::mul(a, st); forms::addc(b, st); } + 43 => { forms::mul(a, st); forms::xorc(b, st); } + 44 => { forms::mul(a, st); forms::mulc(b, st); } + 45 => { forms::mul(a, st); forms::mulc2(b, st); } + 46 => { forms::mul(a, st); forms::andx(b, st); } + 47 => { forms::mul(a, st); forms::orx(b, st); } + 48 => { forms::rot(a, st); forms::add(b, st); } + 49 => { forms::rot(a, st); forms::sub(b, st); } + 50 => { forms::rot(a, st); forms::xor(b, st); } + 51 => { forms::rot(a, st); forms::mul(b, st); } + 52 => { forms::rot(a, st); forms::rot(b, st); } + 53 => { forms::rot(a, st); forms::xrot(b, st); } + 54 => { forms::rot(a, st); forms::addc(b, st); } + 55 => { forms::rot(a, st); forms::xorc(b, st); } + 56 => { forms::rot(a, st); forms::mulc(b, st); } + 57 => { forms::rot(a, st); forms::mulc2(b, st); } + 58 => { forms::rot(a, st); forms::andx(b, st); } + 59 => { forms::rot(a, st); forms::orx(b, st); } + 60 => { forms::xrot(a, st); forms::add(b, st); } + 61 => { forms::xrot(a, st); forms::sub(b, st); } + 62 => { forms::xrot(a, st); forms::xor(b, st); } + 63 => { forms::xrot(a, st); forms::mul(b, st); } + 64 => { forms::xrot(a, st); forms::rot(b, st); } + 65 => { forms::xrot(a, st); forms::xrot(b, st); } + 66 => { forms::xrot(a, st); forms::addc(b, st); } + 67 => { forms::xrot(a, st); forms::xorc(b, st); } + 68 => { forms::xrot(a, st); forms::mulc(b, st); } + 69 => { forms::xrot(a, st); forms::mulc2(b, st); } + 70 => { forms::xrot(a, st); forms::andx(b, st); } + 71 => { forms::xrot(a, st); forms::orx(b, st); } + 72 => { forms::addc(a, st); forms::add(b, st); } + 73 => { forms::addc(a, st); forms::sub(b, st); } + 74 => { forms::addc(a, st); forms::xor(b, st); } + 75 => { forms::addc(a, st); forms::mul(b, st); } + 76 => { forms::addc(a, st); forms::rot(b, st); } + 77 => { forms::addc(a, st); forms::xrot(b, st); } + 78 => { forms::addc(a, st); forms::addc(b, st); } + 79 => { forms::addc(a, st); forms::xorc(b, st); } + 80 => { forms::addc(a, st); forms::mulc(b, st); } + 81 => { forms::addc(a, st); forms::mulc2(b, st); } + 82 => { forms::addc(a, st); forms::andx(b, st); } + 83 => { forms::addc(a, st); forms::orx(b, st); } + 84 => { forms::xorc(a, st); forms::add(b, st); } + 85 => { forms::xorc(a, st); forms::sub(b, st); } + 86 => { forms::xorc(a, st); forms::xor(b, st); } + 87 => { forms::xorc(a, st); forms::mul(b, st); } + 88 => { forms::xorc(a, st); forms::rot(b, st); } + 89 => { forms::xorc(a, st); forms::xrot(b, st); } + 90 => { forms::xorc(a, st); forms::addc(b, st); } + 91 => { forms::xorc(a, st); forms::xorc(b, st); } + 92 => { forms::xorc(a, st); forms::mulc(b, st); } + 93 => { forms::xorc(a, st); forms::mulc2(b, st); } + 94 => { forms::xorc(a, st); forms::andx(b, st); } + 95 => { forms::xorc(a, st); forms::orx(b, st); } + 96 => { forms::mulc(a, st); forms::add(b, st); } + 97 => { forms::mulc(a, st); forms::sub(b, st); } + 98 => { forms::mulc(a, st); forms::xor(b, st); } + 99 => { forms::mulc(a, st); forms::mul(b, st); } + 100 => { forms::mulc(a, st); forms::rot(b, st); } + 101 => { forms::mulc(a, st); forms::xrot(b, st); } + 102 => { forms::mulc(a, st); forms::addc(b, st); } + 103 => { forms::mulc(a, st); forms::xorc(b, st); } + 104 => { forms::mulc(a, st); forms::mulc(b, st); } + 105 => { forms::mulc(a, st); forms::mulc2(b, st); } + 106 => { forms::mulc(a, st); forms::andx(b, st); } + 107 => { forms::mulc(a, st); forms::orx(b, st); } + 108 => { forms::mulc2(a, st); forms::add(b, st); } + 109 => { forms::mulc2(a, st); forms::sub(b, st); } + 110 => { forms::mulc2(a, st); forms::xor(b, st); } + 111 => { forms::mulc2(a, st); forms::mul(b, st); } + 112 => { forms::mulc2(a, st); forms::rot(b, st); } + 113 => { forms::mulc2(a, st); forms::xrot(b, st); } + 114 => { forms::mulc2(a, st); forms::addc(b, st); } + 115 => { forms::mulc2(a, st); forms::xorc(b, st); } + 116 => { forms::mulc2(a, st); forms::mulc(b, st); } + 117 => { forms::mulc2(a, st); forms::mulc2(b, st); } + 118 => { forms::mulc2(a, st); forms::andx(b, st); } + 119 => { forms::mulc2(a, st); forms::orx(b, st); } + 120 => { forms::andx(a, st); forms::add(b, st); } + 121 => { forms::andx(a, st); forms::sub(b, st); } + 122 => { forms::andx(a, st); forms::xor(b, st); } + 123 => { forms::andx(a, st); forms::mul(b, st); } + 124 => { forms::andx(a, st); forms::rot(b, st); } + 125 => { forms::andx(a, st); forms::xrot(b, st); } + 126 => { forms::andx(a, st); forms::addc(b, st); } + 127 => { forms::andx(a, st); forms::xorc(b, st); } + 128 => { forms::andx(a, st); forms::mulc(b, st); } + 129 => { forms::andx(a, st); forms::mulc2(b, st); } + 130 => { forms::andx(a, st); forms::andx(b, st); } + 131 => { forms::andx(a, st); forms::orx(b, st); } + 132 => { forms::orx(a, st); forms::add(b, st); } + 133 => { forms::orx(a, st); forms::sub(b, st); } + 134 => { forms::orx(a, st); forms::xor(b, st); } + 135 => { forms::orx(a, st); forms::mul(b, st); } + 136 => { forms::orx(a, st); forms::rot(b, st); } + 137 => { forms::orx(a, st); forms::xrot(b, st); } + 138 => { forms::orx(a, st); forms::addc(b, st); } + 139 => { forms::orx(a, st); forms::xorc(b, st); } + 140 => { forms::orx(a, st); forms::mulc(b, st); } + 141 => { forms::orx(a, st); forms::mulc2(b, st); } + 142 => { forms::orx(a, st); forms::andx(b, st); } + 143 => { forms::orx(a, st); forms::orx(b, st); } + _ => unreachable!(), + } + } + for ins in it.remainder() { + run_instr(ins, st); + } +} + +/// The scalar reference: one instruction on one 16-word state, the text the kernels carry (`emit.rs`, +/// `derive_instr_text`) restated in Rust. The tests pin the SoA interpreter against it. +pub fn run_round_scalar(prog: &[DInstr], s: &mut [u32; DERIVE_REGS]) { + for ins in prog { + let d = ins.dst as usize; + let c = s[ins.src as usize]; + match ins.op { + DOp::Add => s[d] = s[d].wrapping_add(c), + DOp::Sub => s[d] = s[d].wrapping_sub(c), + DOp::Xor => s[d] ^= c, + DOp::Mul => s[d] = s[d].wrapping_mul(c | 1), + DOp::Rot => s[d] = rotl(s[d], ins.rot as u32).wrapping_add(c), + DOp::XRot => s[d] = rotl(s[d] ^ c, ins.rot as u32), + DOp::AddC => s[d] = s[d].wrapping_add(c.wrapping_add(ins.imm)), + DOp::XorC => s[d] ^= c ^ ins.imm, + DOp::MulC => s[d] = (s[d] ^ c).wrapping_mul(ins.imm), + DOp::MulC2 => s[d] = s[d].wrapping_mul(ins.imm).wrapping_add(c), + DOp::AndX => s[d] ^= c & s[ins.src2 as usize], + DOp::OrX => s[d] = s[d].wrapping_add(c | s[ins.src2 as usize]), + } + } +} + +/// The source text of one instruction in the C-family dialects (the same text in Metal, CUDA C and OpenCL C: +/// `s` is the 16-word state, `mh_rotl` the rotate of the memhard core). +pub fn instr_text(ins: &DInstr) -> String { + let (d, c, b) = (ins.dst, ins.src, ins.src2); + match ins.op { + DOp::Add => format!("s[{d}] += s[{c}];"), + DOp::Sub => format!("s[{d}] -= s[{c}];"), + DOp::Xor => format!("s[{d}] ^= s[{c}];"), + DOp::Mul => format!("s[{d}] *= (s[{c}] | 1u);"), + DOp::Rot => format!("s[{d}] = mh_rotl(s[{d}], {}u) + s[{c}];", ins.rot), + DOp::XRot => format!("s[{d}] = mh_rotl(s[{d}] ^ s[{c}], {}u);", ins.rot), + DOp::AddC => format!("s[{d}] += s[{c}] + {:#010x}u;", ins.imm), + DOp::XorC => format!("s[{d}] ^= s[{c}] ^ {:#010x}u;", ins.imm), + DOp::MulC => format!("s[{d}] = (s[{d}] ^ s[{c}]) * {:#010x}u;", ins.imm), + DOp::MulC2 => format!("s[{d}] = s[{d}] * {:#010x}u + s[{c}];", ins.imm), + DOp::AndX => format!("s[{d}] ^= (s[{c}] & s[{b}]);"), + DOp::OrX => format!("s[{d}] += (s[{c}] | s[{b}]);"), + } +} + +/// One instruction as a line of program.json: `"add d=3 c=0"`, `"mulc d=5 c=3 imm=0x..."`. +pub fn instr_line(ins: &DInstr) -> String { + let mut s = format!("{} d={} c={}", ins.op.name(), ins.dst, ins.src); + if ins.op.has_third() { + s.push_str(&format!(" b={}", ins.src2)); + } + if ins.op.has_rot() { + s.push_str(&format!(" k={}", ins.rot)); + } + if ins.op.has_imm() { + s.push_str(&format!(" imm={:#010x}", ins.imm)); + } + s +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn weights_sum_and_ops() { + assert_eq!(DOP_WEIGHTS.iter().map(|(_, w)| w).sum::(), 100); + for (i, (op, _)) in DOP_WEIGHTS.iter().enumerate() { + assert_eq!(*op as usize, i); + assert_eq!(DOp::from_u8(i as u8), Some(*op)); + } + assert_eq!(DOp::for_roll(0), DOp::Add); + assert_eq!(DOp::for_roll(13), DOp::Add); + assert_eq!(DOp::for_roll(14), DOp::Sub); + assert_eq!(DOp::for_roll(99), DOp::OrX); + // the expected chip operations per instruction, 1.52 (1.62 on a GPU), put 736 x 9 over the x8 floors + let mean: f64 = DOP_WEIGHTS.iter().map(|(o, w)| o.chip_ops() as f64 * *w as f64 / 100.0).sum(); + assert!((mean - 1.52).abs() < 1e-9, "{mean}"); + let gpu_mean: f64 = DOP_WEIGHTS.iter().map(|(o, w)| o.gpu_ops() as f64 * *w as f64 / 100.0).sum(); + assert!((gpu_mean - 1.62).abs() < 1e-9, "{gpu_mean}"); + assert!(9.0 * DERIVE_LEN_X8 as f64 * mean > OPS_FLOOR_X8 as f64); + assert!(9.0 * DERIVE_LEN_X8 as f64 * gpu_mean > GPU_OPS_FLOOR_X8 as f64); + assert_eq!(DeriveProgram::floors(DERIVE_LEN_X8), (9_216, 10_368, 1_152)); + assert_eq!(DeriveProgram::floors(368), (4_608, 5_184, 576)); + let mul_share: f64 = DOP_WEIGHTS.iter().filter(|(o, _)| o.is_mul()).map(|(_, w)| *w as f64 / 100.0).sum(); + assert!(9.0 * DERIVE_LEN_X8 as f64 * mul_share > MULS_FLOOR_X8 as f64); + } + + #[test] + fn draw_is_structural_and_accepted() { + let mut rng = SplitMix64::new(0x1234_5678_9abc_def0); + let p = DeriveProgram::draw(&mut rng, DERIVE_LEN_X8); + assert_eq!(p.attempt, 0, "the first candidate of this seed passes"); + assert_eq!(p.rounds.len(), DERIVE_PROGRAMS); + assert_eq!(p.instr_count(), 9 * DERIVE_LEN_X8 as u64); + assert!(p.check().is_ok()); + assert!(p.chip_ops() >= OPS_FLOOR_X8 && p.gpu_ops() >= GPU_OPS_FLOOR_X8 && p.muls() >= MULS_FLOOR_X8); + assert!(p.gpu_ops() > p.chip_ops()); + // every instruction consumes the newest result + for prog in &p.rounds { + let mut chain = 0u8; + for ins in prog { + assert_eq!(ins.src, chain); + assert_ne!(ins.dst, chain); + chain = ins.dst; + } + } + // four draws per instruction: the same program again from the same seed, and a different one one draw on + let mut rng2 = SplitMix64::new(0x1234_5678_9abc_def0); + assert_eq!(DeriveProgram::draw(&mut rng2, DERIVE_LEN_X8), p); + let mut rng3 = SplitMix64::new(0x1234_5678_9abc_def0); + rng3.next(); + assert_ne!(DeriveProgram::draw(&mut rng3, DERIVE_LEN_X8), p); + } + + #[test] + fn soa_matches_scalar_and_is_a_bijection() { + let mut rng = SplitMix64::new(7); + let p = DeriveProgram::draw(&mut rng, 64); + let mut st: SoaState = [[0u32; SOA_LANES]; DERIVE_REGS]; + let mut scalars = [[0u32; DERIVE_REGS]; SOA_LANES]; + let mut x = SplitMix64::new(99); + for k in 0..SOA_LANES { + for r in 0..DERIVE_REGS { + let v = x.next() as u32; + st[r][k] = v; + scalars[k][r] = v; + } + } + let before = scalars; + for prog in &p.rounds { + run_round(prog, &mut st); + for k in 0..SOA_LANES { + run_round_scalar(prog, &mut scalars[k]); + } + } + for k in 0..SOA_LANES { + for r in 0..DERIVE_REGS { + assert_eq!(st[r][k], scalars[k][r], "lane {k} reg {r}"); + } + } + // distinct inputs stay distinct (a bijection on the state, spot-checked: 32 lanes, no collision) + for a in 0..SOA_LANES { + for b in a + 1..SOA_LANES { + assert_ne!(scalars[a], scalars[b]); + assert_ne!(before[a], before[b]); + } + } + } + + #[test] + fn acceptance_rejects_degenerate_draws() { + let mut rng = SplitMix64::new(3); + let mut p = DeriveProgram::draw(&mut rng, 64); + // a register never written: make every write of round 2 go to the chain's neighbour + let mut q = p.clone(); + for ins in q.rounds[2].iter_mut() { + ins.dst = if ins.src == 1 { 2 } else { 1 }; + } + let mut chain = 0u8; + for ins in q.rounds[2].iter_mut() { + ins.src = chain; + ins.dst = if chain == 1 { 2 } else { 1 }; + chain = ins.dst; + } + assert!(matches!(q.check(), Err(DeriveReject::RegisterNeverWritten { round: 2, .. }))); + // all rotations equal + let mut q = p.clone(); + for ins in q.rounds.iter_mut().flatten() { + if ins.op.has_rot() { + ins.rot = 5; + } + } + assert!(matches!(q.check(), Err(DeriveReject::RotationsDegenerate { distinct: 1 }))); + // every op an add apart from the xor-rotates (so the rotations stay distinct): under the ops floor + for ins in p.rounds.iter_mut().flatten() { + if ins.op != DOp::XRot { + ins.op = DOp::Add; + ins.rot = 0; + ins.imm = 0; + ins.src2 = 0; + } + } + assert!(matches!(p.check(), Err(DeriveReject::OpsUnderFloor { .. })), "{:?}", p.check()); + // no multiplies at all but the ops floor met: under the multiply floor + let mut q = DeriveProgram::draw(&mut SplitMix64::new(11), 64); + for ins in q.rounds.iter_mut().flatten() { + if ins.op.is_mul() { + ins.op = DOp::AddC; + } + } + assert!(matches!(q.check(), Err(DeriveReject::MulsUnderFloor { .. })), "{:?}", q.check()); + } + + #[test] + fn text_forms() { + let i = DInstr { op: DOp::MulC, dst: 5, src: 3, src2: 0, rot: 0, imm: 0x9e37_79b9 }; + assert_eq!(instr_text(&i), "s[5] = (s[5] ^ s[3]) * 0x9e3779b9u;"); + assert_eq!(instr_line(&i), "mulc d=5 c=3 imm=0x9e3779b9"); + let i = DInstr { op: DOp::XRot, dst: 0, src: 15, src2: 0, rot: 17, imm: 0 }; + assert_eq!(instr_text(&i), "s[0] = mh_rotl(s[0] ^ s[15], 17u);"); + let i = DInstr { op: DOp::AndX, dst: 2, src: 9, src2: 14, rot: 0, imm: 0 }; + assert_eq!(instr_text(&i), "s[2] ^= (s[9] & s[14]);"); + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/emit.rs b/tools/attack/adv-accept-v5/igneum-pow/src/emit.rs new file mode 100644 index 000000000..90105ba01 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/emit.rs @@ -0,0 +1,2316 @@ +//! Kernel source emitters. Since 4 October 2026 (generator version 2) this crate is the source of every pack in +//! `proto-cuda/packs/`; the pack tests diff the emitters against the checked-in files. Each function started as a +//! byte-for-byte twin of its namesake in `proto-metal/main.swift` (`generateMSL`, `memhardMSL`, `emitMemhardCore`, +//! `generateCUDA`, `generateOpenCL`, `generateProgramHeader`, `generateMemhardHeader`, `generateVectorsHeader`, +//! `generateProgramJSON`, `generateVectorsJSON`); the kernel text is unchanged by version 2, and `program.json` and +//! `program.h` carry the generator version, the attempt and the program id so no version 1 pack can be mistaken +//! for a current one. +//! +//! One deliberate difference from the Swift: `program_json` writes the cache line mask inside the `"item"` string +//! as a bare `0x003fffff`. The Swift writes it quoted (`jhex`), which is not valid JSON. + +use crate::generator::{EraParams, Instr, Op, Program, ProgramClass, GENERATOR_VERSION, INSTR_COUNT, ITERATIONS, LOAD_SLOTS, PROGRAM_SUBVERSION_V4}; +use crate::derive::{instr_line as derive_instr_line, instr_text as derive_instr_text, DERIVE_PROGRAMS}; +use crate::memhard::{ + hot_key, hot_segments, hot_words, Layout, MixParams, Shape, CACHE_LINES_PER_SEGMENT, CACHE_SEGMENT_LOG2_LINES, CACHE_TAG, + CHACHA_ROUNDS, CHACHA_SIGMA, HOT_TAG, ITEM_ROUNDS, +}; +use crate::seed::SplitMix64; +use crate::verify::{window, DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2, FOLD_MUL, FOLD_ROT}; + +/// The index expression of a dataset load (era layout, `docs/plans/era-layout.md` section 1.3). For every class +/// without an era it is the lottery hash's `rN & MASK`; for an era program it is the one form +/// `((rotl_imm(rN * M, R) & WM) | OFF) & MASK` with the site's window constants at the pack's dataset size. +fn load_index_expr(dialect: CoreDialect, era: Option<&EraParams>, ins: &Instr, a: &str, dataset_log2: u32) -> String { + let mask_name = match dialect { + CoreDialect::Metal => "MASK", + _ => "mask", + }; + match era { + None => format!("{a} & {mask_name}"), + Some(e) => { + let (wm, off) = window(ins, mask_for(dataset_log2), dataset_log2); + format!("((rotl_imm({a} * {}, {}u) & {}) | {}) & {mask_name}", hex(e.stride_mul), e.stride_rot, hex(wm), hex(off)) + } + } +} + +/// The era lines of program.h (empty without an era). +fn era_header_lines(p: &Program) -> String { + let Some(e) = p.class.era else { return String::new() }; + let mut s = String::new(); + s.push_str("// Era layout (5 October 2026, docs/plans/era-layout.md): NOT the lottery hash. Every dataset load reads\n"); + s.push_str("// idx = ((rotl(src * STRIDE_MUL, STRIDE_ROT) & window mask) | window offset) & MASK; the window of a load site is the\n"); + s.push_str("// dataset, a half or a quarter of it (IGNEUM_ERA_WINDOWS: site:shrink:offset); dataset word w holds word j(w) of item\n"); + s.push_str("// t(w) with j's bits at the INTERLEAVE positions (memhard.h: mh_t, mh_j, mh_addr).\n"); + s.push_str(&format!("#define IGNEUM_ERA_LABEL {}\n", jstr(&e.label()))); + s.push_str(&format!("#define IGNEUM_ERA_SEED_WORDS {{ {} }}\n", join_hex(&e.words))); + s.push_str(&format!("#define IGNEUM_ERA_ALLOWED_WIDTHS {{ {}, {}, {} }} // words, ascending, 0 = unused; one entry pins the width\n", e.allowed[0], e.allowed[1], e.allowed[2])); + s.push_str(&format!("#define IGNEUM_ERA_WIDTH_WORDS {}\n", e.width_words)); + s.push_str(&format!("#define IGNEUM_ERA_STRIDE_MUL {}\n", hex(e.stride_mul))); + s.push_str(&format!("#define IGNEUM_ERA_STRIDE_ROT {}\n", e.stride_rot)); + s.push_str(&format!("#define IGNEUM_ERA_INTERLEAVE {{ {}, {}, {}, {} }}\n", e.pos[0], e.pos[1], e.pos[2], e.pos[3])); + s.push_str(&format!("#define IGNEUM_ERA_WINDOWS {}\n", jstr(&era_windows(p)))); + s +} + +/// "site:shrink:offset" for every load site of an era program, space separated. +fn era_windows(p: &Program) -> String { + p.instrs + .iter() + .enumerate() + .filter(|(_, i)| i.op == Op::Load) + .map(|(k, i)| format!("{k}:{}:{}", i.win, i.off)) + .collect::>() + .join(" ") +} + +/// The layout helpers of the memory-hard core for a non-linear layout: `mh_j(w)`, `mh_t(w)` and `mh_addr(t, j)` +/// (`Layout::split` and `Layout::join` as text). Empty for the linear layout, so the pinned packs do not change. +fn layout_helpers(layout: Layout, u: &str, fn_: &str) -> String { + if layout.is_linear() { + return String::new(); + } + let p = layout.pos; + let low = |q: u8| hex(((1u64 << q) - 1) as u32); + let mut s = String::new(); + s.push_str(&format!( + "// Era layout (docs/plans/era-layout.md 1.2): dataset word w holds word mh_j(w) of item mh_t(w); j's bits sit at positions {} {} {} {} of w.\n", + p[0], p[1], p[2], p[3] + )); + s.push_str(&format!( + "{fn_} {u} mh_j({u} w) {{ return ((w >> {}u) & 1u) | (((w >> {}u) & 1u) << 1) | (((w >> {}u) & 1u) << 2) | (((w >> {}u) & 1u) << 3); }}\n", + p[0], p[1], p[2], p[3] + )); + s.push_str(&format!("{fn_} {u} mh_t({u} w) {{")); + for &q in p.iter().rev() { + s.push_str(&format!(" w = (w & {}) | ((w >> {}u) << {}u);", low(q), q + 1, q)); + } + s.push_str(" return w; }\n"); + s.push_str(&format!("{fn_} {u} mh_addr({u} t, {u} j) {{ {u} w = t;")); + for (i, &q) in p.iter().enumerate() { + s.push_str(&format!(" w = ((w >> {}u) << {}u) | (w & {}) | (((j >> {}u) & 1u) << {}u);", q, q + 1, low(q), i, q)); + } + s.push_str(" return w; }\n"); + s +} + +/// Where the words of a wide load come from (read-width experiment). +#[derive(Clone, Copy, PartialEq, Eq)] +enum WideSource { + /// `dataset`/`ds`: vector loads from the stored dataset. + Stored, + /// the closed form per word (Metal inline shortcut kernel). + InlineClosed, + /// one `mh_item` derivation per load, words taken from it (Metal inline memory-hard kernel). + InlineMemhard, +} + +/// One wide `load` as a single statement block (read-width experiment, 5 October 2026): `width` words from the +/// address aligned down to `width` words, folded into `dst` as `verify::fold_words`. The emitted text is the +/// same shape in the three dialects: the vector loads differ (`uint4` pointer on Metal and CUDA, `vload4` on +/// OpenCL C 1.2). For `width == 1` the caller emits the lottery hash's one-word form instead. +fn wide_load_stmt(dialect: CoreDialect, d: &str, idx: &str, width: u8, src: WideSource, closed: Option<(u32, u32)>) -> String { + debug_assert!(width == 4 || width == 16); + let (u, base_ptr) = match dialect { + CoreDialect::Metal => ("uint", "dataset"), + CoreDialect::Cuda => ("uint32_t", "ds"), + CoreDialect::OpenCl => ("uint", "ds"), + }; + let vectors = width as usize / 4; + let mut s = String::with_capacity(400); + s.push_str(&format!("{{ {u} b_ = ({idx}) & ~{}u; ", width as u32 - 1)); + match src { + WideSource::Stored => match dialect { + CoreDialect::Metal => s.push_str(&format!("device const uint4* l_ = (device const uint4*)({base_ptr} + b_); ")), + CoreDialect::Cuda => s.push_str(&format!("const uint4* l_ = (const uint4*)({base_ptr} + b_); ")), + CoreDialect::OpenCl => {} + }, + WideSource::InlineClosed => {} + WideSource::InlineMemhard => s.push_str("uint s_[16]; mh_item(cache, mh_t(b_), s_); "), + } + let word = |j: usize| -> String { + match src { + WideSource::Stored => format!("v{}_.{}", j / 4, ["x", "y", "z", "w"][j % 4]), + WideSource::InlineClosed => { + let (d0, d1) = closed.expect("closed-form words need d0, d1"); + format!("ds_elem(b_ + {j}u, {}, {})", hex(d0), hex(d1)) + } + WideSource::InlineMemhard => format!("s_[mh_j(b_) + {j}u]"), + } + }; + if src == WideSource::Stored { + for v in 0..vectors { + match dialect { + CoreDialect::OpenCl => s.push_str(&format!("uint4 v{v}_ = vload4({v}u, {base_ptr} + b_); ")), + _ => s.push_str(&format!("uint4 v{v}_ = l_[{v}]; ")), + } + } + } + s.push_str(&format!("{u} x_ = {d} ^ {}; ", word(0))); + for j in 1..width as usize { + s.push_str(&format!("x_ = (rotl_imm(x_, {FOLD_ROT}u) * {}) ^ {}; ", hex(FOLD_MUL), word(j))); + } + s.push_str(&format!("{d} = x_; }}")); + s +} + +/// The program class lines of program.h (Counter ASIC 2.0): `IGNEUM_PROGRAM_CLASS` and, when the program was drawn +/// on the chain, `IGNEUM_ERA_SEED_HEX`. Empty for every version 2 program, so the pinned packs do not change; a +/// worker reads an absent line as class v2. The generator version is `IGNEUM_GENERATOR` as before (3 for class v3, +/// 4 for class v4; the v3 comment lines are byte for byte the 0.3.11 ones, so the pinned v3 packs do not change). +fn program_class_header_lines(p: &Program) -> String { + if p.program_class() == ProgramClass::V2 { + return String::new(); + } + let mut s = String::new(); + if p.program_class() == ProgramClass::V5 { + s.push_str("// Program class v5 (proof of stored state and of following, docs/design/class-v5-stored-state.md): generator version 5,\n"); + s.push_str("// class v4 over a dataset whose every item is keyed by the window's execution state (IGNEUM_STATE_* below, leaves.bin);\n"); + s.push_str("// a worker that runs another class refuses this pack, and a job line names the class it wants (class=v5 era=).\n"); + } else if p.program_class() == ProgramClass::V4 { + s.push_str("// Program class v4 (Counter ASIC 3.0, docs/plans/counter-asic-3-node.md): generator version 4, class v3 plus the\n"); + s.push_str("// latency-shadow block (IGNEUM_SHADOW_INSTRS x IGNEUM_SHADOW_REPS per iteration); a worker that runs another class\n"); + s.push_str("// refuses this pack, and a job line names the class it wants (class=v4 era=).\n"); + } else { + s.push_str("// Program class v3 (Counter ASIC 2.0, docs/plans/counter-asic-2-rollout.md): generator version 3; a worker that\n"); + s.push_str("// runs another class refuses this pack, and a job line names the class it wants (class=v3 era=).\n"); + } + s.push_str(&format!("#define IGNEUM_PROGRAM_CLASS {}\n", jstr(p.program_class().name()))); + if p.program_class() == ProgramClass::V4 { + // the class v4 stream sub-version (AP-F8-1 amendment): a worker ignores it, packcheck requires it + s.push_str(&format!("#define IGNEUM_PROGRAM_SUBVERSION {}\n", PROGRAM_SUBVERSION_V4)); + } + if let Some(era) = &p.era_bytes { + s.push_str(&format!("#define IGNEUM_ERA_SEED_HEX {}\n", jstr(&hex_bytes(era)))); + } + s +} + +/// The state lines of program.h (class v5): the window's reference block and state root, the leaf count, the FNV of +/// `leaves.bin` and the file's name. Empty for every dataset without leaves, so no pinned pack changes. +fn state_header_lines(ds: &DatasetSource) -> String { + let Some(l) = ds.leaves() else { return String::new() }; + let mut s = String::new(); + s.push_str("// Class v5 state (docs/design/class-v5-stored-state.md): the window's reference chain block and the state root after it;\n"); + s.push_str("// leaves.bin holds IGNEUM_STATE_LEAVES leaves of 16 little-endian words, leaf(t) = leaves[t mod IGNEUM_STATE_LEAVES].\n"); + s.push_str(&format!("#define IGNEUM_STATE_BLOCK_HEX {}\n", jstr(&hex_bytes(&l.block)))); + s.push_str(&format!("#define IGNEUM_STATE_BLOCK_NUMBER {}\n", l.number)); + s.push_str(&format!("#define IGNEUM_STATE_ROOT_HEX {}\n", jstr(&hex_bytes(&l.root)))); + s.push_str(&format!("#define IGNEUM_STATE_LEAVES {}\n", l.n())); + s.push_str(&format!("#define IGNEUM_STATE_RECORDS {}\n", l.records_total)); + s.push_str(&format!("#define IGNEUM_STATE_SAMPLED {}\n", l.sampled as u8)); + s.push_str(&format!("#define IGNEUM_STATE_LEAVES_FNV64 {}\n", hex64(l.fnv1a64()))); + s.push_str("#define IGNEUM_STATE_LEAVES_FILE \"leaves.bin\"\n"); + s +} + +/// The load class lines of program.h (empty for the lottery hash, so the pinned packs do not change). +fn class_header_lines(p: &Program) -> String { + if p.class.is_v2() { + return String::new(); + } + let mut s = String::new(); + if p.class.v2_loads() { + s.push_str("// Class v3 construction (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): version 2 loads; the dataset item +"); + s.push_str("// derivation applies the mixer IGNEUM_MIXER_MULT times per round (memhard.h), and the cache follows the growth rule. +"); + } else { + s.push_str("// Read-width experiment (5 October 2026, docs/plans/read-width.md): NOT the lottery hash. A load of W words reads +"); + s.push_str("// the W-word-aligned address and folds every word into dst: x = dst ^ w[0]; x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x. +"); + } + s.push_str(&format!("#define IGNEUM_LOAD_CLASS {} +", jstr(&p.class.name()))); + if p.class.mixer_mult != 1 || p.class.growth { + s.push_str(&format!("#define IGNEUM_CLASS_MIXER_MULT {} +", p.class.mixer_mult)); + s.push_str(&format!("#define IGNEUM_CACHE_GROWTH {} // 1: cache words = 2^(26 + doublings(day)), doublings = floor(log2(1 + day / 1460)) +", p.class.growth as u8)); + } + if p.class.derive_len != 0 { + s.push_str("// Counter ASIC 3.0 item 2 (6 October 2026, docs/plans/counter-asic-3-derivation.md, a prototype, NOT class v3): the item +"); + s.push_str("// derivation runs the day's drawn program (memhard.h: mh_round_0..8, IGNEUM_DERIVE_LEN instructions each) in place of the mixer. +"); + s.push_str(&format!("#define IGNEUM_CLASS_DERIVE_LEN {} +", p.class.derive_len)); + } + s.push_str(&format!("#define IGNEUM_LOAD_SLOTS {} +", p.class.load_slots)); + s.push_str(&format!("#define IGNEUM_LOAD_MIX {{ {}, {}, {} }} +", p.class.mix[0], p.class.mix[1], p.class.mix[2])); + let c = p.width_counts(); + s.push_str(&format!("#define IGNEUM_LOAD_WIDTH_COUNTS {{ {}, {}, {} }} // loads of 4, 16, 64 bytes per program +", c[0], c[1], c[2])); + s.push_str(&format!("#define IGNEUM_BYTES_PER_HASH {} +", p.bytes_per_hash())); + s.push_str(&format!("#define IGNEUM_FOLD_ROT {FOLD_ROT} +")); + s.push_str(&format!("#define IGNEUM_FOLD_MUL {} +", hex(FOLD_MUL))); + s +} + +/// The width of a load instruction's statement, for the emitters (1 for every non-load op). +fn load_width(ins: &Instr) -> u8 { + if ins.op == Op::Load { + ins.width + } else { + 1 + } +} + +/// Variant 5 prelude: `scr_fill(gbase, lane, slot, j)`, the fill word of a scratch slot (`verify::scratch_fill`), +/// with the program's seed words as literals. +fn scratch_prelude(p: &Program, dialect: CoreDialect) -> String { + if !p.has_scratch() { + return String::new(); + } + let (u, fn_) = match dialect { + CoreDialect::Metal => ("uint", "inline"), + CoreDialect::Cuda => ("uint32_t", "__device__ __forceinline__"), + CoreDialect::OpenCl => ("uint", "static inline"), + }; + let mut s = String::new(); + s.push_str(&format!("// Variant 5 (read-width experiment, 5 October 2026, NOT the lottery hash): a {} KiB scratch per warp, {} slots of\n", p.class.scratch_kb, p.class.scratch_slots_per_lane())); + s.push_str("// 16 bytes per lane, lane-major. A slot starts the unit as the fill words below (tagged lazily: a slot whose tag is not\n"); + s.push_str("// this unit's reads as its fill) and holds what the unit wrote afterwards. scr_fill mirrors verify::scratch_fill.\n"); + s.push_str(&format!( + "{fn_} {u} scr_fill({u} gbase, {u} lane, {u} slot, {u} j) {{ {u} sw = (j == 0u) ? {} : ((j == 1u) ? {} : {}); return splitmix32(((gbase + lane) ^ sw) + slot * 0x9e3779b1u + (j + 1u) * 0x85ebca77u); }}\n", + hex(p.seed[0]), + hex(p.seed[1]), + hex(p.seed[2]) + )); + s +} + +/// Variant 5: one scratch read-modify-write as a statement block. `arena`, `tag`, `gbase` and `lane` are in scope +/// (the persistent prologue). Reads 16 bytes, folds the three data words into dst, rewrites the slot behind the tag. +fn scratch_stmt(dialect: CoreDialect, d: &str, a: &str, slot_mask: u32) -> String { + let (u, load, store) = match dialect { + CoreDialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"), + CoreDialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_);"), + CoreDialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena);"), + }; + format!( + "{{ {u} s_ = {a} & {slot_mask}u; {load} {u} m_ = (v_.x == tag) ? 0xffffffffu : 0u; {u} w0_ = (v_.y & m_) | (scr_fill(gbase, lane, s_, 0u) & ~m_); {u} w1_ = (v_.z & m_) | (scr_fill(gbase, lane, s_, 1u) & ~m_); {u} w2_ = (v_.w & m_) | (scr_fill(gbase, lane, s_, 2u) & ~m_); {u} x_ = {d} ^ w0_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w1_; x_ = (rotl_imm(x_, {FOLD_ROT}u) * {k}) ^ w2_; {d} = x_; {store} }}", + k = hex(FOLD_MUL) + ) +} + +/// Variant 5: the persistent-warp prologue. The kernel is launched with N warps (the resident count, the host's +/// choice); warp `w` owns arena `w` and runs the units `w, w + N, w + 2N, ...` of the launch. Inside the loop the +/// lottery hash's text is unchanged: `gid` is the unit's first output index plus the lane. The host MUST launch +/// `groups` as a multiple of N (a uniform trip count: the OpenCL local-memory exchange carries a barrier). +fn persistent_prologue(dialect: CoreDialect, words_per_lane: usize) -> String { + let (a, b) = persistent_prologue_parts(dialect, words_per_lane); + a + &b +} + +/// The prologue in two parts: the warp's identity and arena, then the unit loop. OpenCL C requires a `__local` +/// variable at the outermost scope of the kernel (AMD's compiler enforces it, 5 October 2026, round 3 on the +/// 9070 XT), so the OpenCL kernels declare the exchange buffer between the two parts. +fn persistent_prologue_parts(dialect: CoreDialect, words_per_lane: usize) -> (String, String) { + let (u, tid, nthreads, ptr) = match dialect { + CoreDialect::Metal => ("uint", "tid", "nthreads", "device uint*"), + CoreDialect::Cuda => ("uint32_t", "(blockIdx.x * blockDim.x + threadIdx.x)", "(gridDim.x * blockDim.x)", "uint32_t*"), + CoreDialect::OpenCl => ("uint", "(uint)get_global_id(0)", "(uint)get_global_size(0)", "__global uint*"), + }; + let mut s = String::new(); + s.push_str(&format!(" {u} lane = {tid} & 31u;\n")); + s.push_str(&format!(" {u} warp_ = {tid} >> 5;\n")); + s.push_str(&format!(" {u} nwarps_ = {nthreads} >> 5;\n")); + s.push_str(&format!(" {ptr} arena = scratch + ((size_t)warp_ * 32u + lane) * {words_per_lane}u;\n")); + let mut l = String::new(); + l.push_str(&format!(" for ({u} g_ = warp_; g_ < groups; g_ += nwarps_) {{\n")); + l.push_str(&format!(" {u} gid = g_ * 32u + lane;\n")); + l.push_str(&format!(" {u} gbase = baseNonce + g_ * 32u;\n")); + l.push_str(&format!(" {u} tag = salt + g_;\n")); + (s, l) +} + +/// The scratch lines of program.h (variant 5). +fn scratch_header_lines(p: &Program) -> String { + if !p.has_scratch() { + return String::new(); + } + let mut s = String::new(); + s.push_str(&format!("// Variant 5: persistent warps, a {} KiB scratch per launched warp (the host launches N warps and passes scratch,\n", p.class.scratch_kb)); + s.push_str("// groups and salt as the last three kernel arguments; groups must be a multiple of N; the tag of a unit is salt + unit).\n"); + s.push_str("#define IGNEUM_PERSISTENT_WARPS 1\n"); + s.push_str(&format!("#define IGNEUM_SCRATCH_OPS {} // scratch read-modify-writes per program ({} per hash)\n", p.class.scratch_slots(), p.scratch_ops_per_hash())); + s.push_str(&format!("#define IGNEUM_SCRATCH_SLOTS {}u\n", p.class.scratch_slots_per_lane())); + s.push_str(&format!("#define IGNEUM_SCRATCH_WORDS_PER_LANE {}u\n", p.class.scratch_words_per_lane())); + s.push_str(&format!("#define IGNEUM_SCRATCH_BYTES_PER_WARP {}u\n", p.class.scratch_bytes_per_warp())); + s +} + +/// Hot-table experiment (`docs/plans/hot-table.md`): the `HOT_WORDS` literal of a hot pack's hash kernels (empty +/// for every other class, so the pinned packs do not change). +/// One ALU instruction of the latency-shadow block as a statement of the dialect (the same text the program's own +/// instruction lines use for that op; the block holds no load, scratch, hot or wide op). +fn shadow_instr_line(dialect: CoreDialect, ins: &Instr) -> String { + let d = format!("r{}", ins.dst); + let a = format!("r{}", ins.src); + let b = format!("r{}", ins.src2); + match ins.op { + Op::Add => match dialect { + CoreDialect::Metal => format!("{d} = {d} + {a} + select({}, {}, ((sel >> {}u) & 1u) != 0u);", hex(ins.imm), hex(ins.imm2), ins.bit), + _ => format!("{d} = {d} + {a} + ((((sel >> {}u) & 1u) != 0u) ? {} : {});", ins.bit, hex(ins.imm2), hex(ins.imm)), + }, + Op::Sub => format!("{d} = {d} - {a};"), + Op::Mul => format!("{d} = {d} * {a};"), + Op::MulHi => match dialect { + CoreDialect::Metal => format!("{d} = mulhi({d}, {a});"), + CoreDialect::Cuda => format!("{d} = __umulhi({d}, {a});"), + CoreDialect::OpenCl => format!("{d} = mul_hi({d}, {a});"), + }, + Op::Xor => format!("{d} = {d} ^ {a};"), + Op::Or => format!("{d} = {d} | {a};"), + Op::Rotl => format!("{d} = rotl_imm({d}, {}u);", ins.rot), + Op::Rotr => format!("{d} = rotr_var({d}, {a});"), + Op::Mad => format!("{d} = {a} * {b} + {d};"), + Op::Shfl => match dialect { + CoreDialect::Metal => format!("{d} = {d} ^ simd_shuffle_xor({a}, (ushort){});", ins.mask), + CoreDialect::Cuda => format!("{d} = {d} ^ __shfl_xor_sync(0xffffffffu, {a}, {});", ins.mask), + CoreDialect::OpenCl => format!("{{ uint t_; IGNEUM_SHFL_XOR(t_, {a}, {}u); {d} = {d} ^ t_; }}", ins.mask), + }, + Op::Load | Op::WLoad | Op::Scratch | Op::Hot => unreachable!("the shadow block holds ALU instructions only"), + } +} + +/// The latency-shadow block (Counter ASIC 3.0 item 8, `docs/analysis/latency-shadow-2026-10-06.md`) inside the +/// iteration loop, after the program's last instruction: `reps` passes over the block with the iteration's `sel`. +/// Empty for every program without a shadow, so the pinned v2 and v3 packs do not change by a byte. +fn shadow_block(p: &Program, dialect: CoreDialect) -> String { + if !p.has_shadow() { + return String::new(); + } + let reps = p.shadow_reps(); + let mut s = String::with_capacity(64 * p.shadow.len() + 200); + s.push_str(&format!( + " // latency-shadow block (Counter ASIC 3.0 item 8): {} ALU instructions x {reps} passes after instruction 63, no load\n", + p.shadow.len() + )); + let ty = if dialect == CoreDialect::Cuda { "uint32_t" } else { "uint" }; + s.push_str(&format!(" for ({ty} sh = 0u; sh < {reps}u; ++sh) {{\n")); + for (k, ins) in p.shadow.iter().enumerate() { + s.push_str(&format!(" {} // s{k} {}\n", shadow_instr_line(dialect, ins), ins.op.name())); + } + s.push_str(" }\n"); + s +} + +/// The shadow lines of program.h (empty without a shadow). +fn shadow_header_lines(p: &Program) -> String { + let Some(sh) = p.class.shadow else { return String::new() }; + let mut s = String::new(); + s.push_str("// Latency-shadow block (Counter ASIC 3.0 item 8, docs/analysis/latency-shadow-2026-10-06.md): NOT the lottery hash. A block of\n"); + s.push_str("// IGNEUM_SHADOW_INSTRS ALU instructions (the ten non-load families) runs IGNEUM_SHADOW_REPS times at the end of every\n"); + s.push_str("// iteration; the 16 loads, the acceptance rule and the base program are the class's without the shadow.\n"); + s.push_str(&format!("#define IGNEUM_SHADOW_INSTRS {}\n", sh.instrs)); + s.push_str(&format!("#define IGNEUM_SHADOW_REPS {}\n", sh.reps)); + s.push_str(&format!("#define IGNEUM_SHADOW_INSTRS_PER_HASH {}\n", p.shadow_instrs_per_hash())); + s.push_str(&format!("#define IGNEUM_SHADOW_OP_MIX {}\n", jstr(&p.shadow_op_mix()))); + s +} + +fn hot_define(p: &Program) -> String { + match p.class.hot { + Some(h) => format!("// Hot table ({} MiB, {} of the 16 load slots): dst ^= hot[mulhi(src, HOT_WORDS)], docs/plans/hot-table.md. +#define HOT_WORDS {} +", h.mb, h.k, hex(hot_words(h.mb as u32))), + None => String::new(), + } +} + +/// The hot load as one statement per dialect: the high 32 bits of `src x HOT_WORDS` index the table. +fn hot_stmt(dialect: CoreDialect, d: &str, a: &str) -> String { + match dialect { + CoreDialect::Metal => format!("{d} = {d} ^ hot[mulhi({a}, HOT_WORDS)];"), + CoreDialect::Cuda => format!("{d} = {d} ^ hot[__umulhi({a}, HOT_WORDS)];"), + CoreDialect::OpenCl => format!("{d} = {d} ^ hot[mul_hi({a}, HOT_WORDS)];"), + } +} + +/// The hot table's fill core: `ht_segment(hot, seg)`, the cache chain of `mh_cache_segment` under the hot key and +/// the hot tag (`mh_chacha_block` and `MH_SEGMENT_LINES` must be in scope: the memory-hard core comes first). +fn emit_hot_core(p: &Program, dialect: CoreDialect) -> String { + let Some(h) = p.class.hot else { return String::new() }; + let (u, fn_, wptr) = match dialect { + CoreDialect::Metal => ("uint", "inline", "device uint*"), + CoreDialect::Cuda => ("uint32_t", "IGNEUM_HD", "uint32_t*"), + CoreDialect::OpenCl => ("uint", "static inline", "__global uint*"), + }; + let k = hot_key(&p.seed_bytes); + let mut s = String::with_capacity(1500); + s.push_str(&format!( + "// Hot table (docs/plans/hot-table.md): {} MiB = {} segments of {} chained ChaCha{} lines under the hot key KH = seed_words(\"igneum-hot/\" || epoch seed bytes), tag \"Igne\" \"umHT\". The cache chain with another key and tag.\n", + h.mb, + hot_segments(h.mb as u32), + CACHE_LINES_PER_SEGMENT, + CHACHA_ROUNDS + )); + s.push_str(&format!("{fn_} void ht_segment({wptr} hot, {u} seg) {{\n")); + s.push_str(&format!(" {u} prev[16]; {u} x[16]; {u} y[16];\n")); + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) prev[i] = 0u;\n")); + s.push_str(&format!(" for ({u} j = 0u; j < MH_SEGMENT_LINES; ++j) {{\n")); + s.push_str(&format!( + " x[0] = {} ^ prev[0]; x[1] = {} ^ prev[1]; x[2] = {} ^ prev[2]; x[3] = {} ^ prev[3];\n", + hex(CHACHA_SIGMA[0]), + hex(CHACHA_SIGMA[1]), + hex(CHACHA_SIGMA[2]), + hex(CHACHA_SIGMA[3]) + )); + for i in 0..8 { + s.push_str(&format!(" x[{}] = {} ^ prev[{}];\n", 4 + i, hex(k[i]), 4 + i)); + } + s.push_str(&format!( + " x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = {} ^ prev[14]; x[15] = {} ^ prev[15];\n", + hex(HOT_TAG[0]), + hex(HOT_TAG[1]) + )); + s.push_str(" mh_chacha_block(x, y);\n"); + s.push_str(&format!(" {wptr} line = hot + ((seg * MH_SEGMENT_LINES + j) * 16u);\n")); + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) {{ line[i] = y[i]; prev[i] = y[i]; }}\n")); + s.push_str(" }\n"); + s.push_str("}\n"); + s +} + +/// The hot lines of program.h. +fn hot_header_lines(p: &Program) -> String { + let Some(h) = p.class.hot else { return String::new() }; + let mut s = String::new(); + s.push_str("// Hot table (hot-table experiment, 5 October 2026, docs/plans/hot-table.md): NOT the lottery hash. IGNEUM_HOT_SLOTS of the\n"); + s.push_str("// 16 load slots read H[mulhi(src, IGNEUM_HOT_WORDS)], H = IGNEUM_HOT_MB MiB of chained ChaCha12 lines (the cache chain) under\n"); + s.push_str("// KH = seed_words(\"igneum-hot/\" || epoch seed bytes) and the tag \"Igne\" \"umHT\"; filled by igneum_hot_fill once per epoch.\n"); + s.push_str(&format!("#define IGNEUM_HOT_MB {}\n", h.mb)); + s.push_str(&format!("#define IGNEUM_HOT_WORDS {}\n", hex(hot_words(h.mb as u32)))); + s.push_str(&format!("#define IGNEUM_HOT_SEGMENTS {}u\n", hot_segments(h.mb as u32))); + s.push_str(&format!("#define IGNEUM_HOT_SLOTS {} // hot loads per program ({} per hash), {} the dataset loads ({} of them)\n", h.k, p.hot_loads_per_hash(), if h.added { "added beside" } else { "replacing" }, p.class.dataset_slots())); + s.push_str(&format!("#define IGNEUM_HOT_ADDED {}\n", h.added as u8)); + s.push_str(&format!("#define IGNEUM_HOT_KEY_INIT {{ {} }}\n", join_hex(&hot_key(&p.seed_bytes)))); + s +} + +pub fn hex(v: u32) -> String { + format!("0x{v:08x}u") +} +pub fn hex64(v: u64) -> String { + format!("0x{v:016x}ull") +} +fn jhex(v: u32) -> String { + format!("\"0x{v:08x}\"") +} +fn jhex64(v: u64) -> String { + format!("\"0x{v:016x}\"") +} +/// JSON string with the three escapes the Swift applies (quote, backslash, newline). +fn jstr(s: &str) -> String { + let mut o = String::with_capacity(s.len() + 2); + o.push('"'); + for c in s.chars() { + match c { + '"' => o.push_str("\\\""), + '\\' => o.push_str("\\\\"), + '\n' => o.push_str("\\n"), + _ => o.push(c), + } + } + o.push('"'); + o +} +fn join_hex(v: &[u32]) -> String { + v.iter().map(|&x| hex(x)).collect::>().join(", ") +} +fn join_jhex(v: &[u32]) -> String { + v.iter().map(|&x| jhex(x)).collect::>().join(", ") +} + +/// The three dialects of the memory-hard core. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum CoreDialect { + Metal, + Cuda, + OpenCl, +} + +/// How the hash kernel gets dataset words. `Stored` is the honest kernel; the inline variants are the +/// shortcut measurements of MEMHARD.md section 2.2. +pub enum LoadSource<'a> { + Stored, + InlineClosed(u32, u32), + InlineMemhard(&'a MixParams), +} + +fn log2_segments(shape: &Shape) -> usize { + shape.log2_segments() as usize +} + +/// The memory-hard core as source text (`emitMemhardCore`). Every parameter is a literal. Linear layout. +pub fn emit_memhard_core(mp: &MixParams, dialect: CoreDialect) -> String { + emit_memhard_core_layout(mp, dialect, Layout::LINEAR) +} + +/// [`emit_memhard_core`] with the dataset layout: with a non-linear layout `mh_word` and the build kernels go +/// through `mh_t`, `mh_j` and `mh_addr` (era layout); with the linear layout the text is unchanged. +pub fn emit_memhard_core_layout(mp: &MixParams, dialect: CoreDialect, layout: Layout) -> String { + let shape = &mp.shape; + let m = shape.mixer_mult; + let cache_log2_words = shape.cache_log2_words; + let cache_line_mask = shape.cache_line_mask(); + let (u, fn_, cptr, wptr, lptr, lcptr) = match dialect { + CoreDialect::Metal => { + ("uint", "inline", "device const uint*", "device uint*", "thread uint*", "const thread uint*") + } + CoreDialect::Cuda => ("uint32_t", "IGNEUM_HD", "const uint32_t*", "uint32_t*", "uint32_t*", "const uint32_t*"), + CoreDialect::OpenCl => { + ("uint", "static inline", "__global const uint*", "__global uint*", "uint*", "const uint*") + } + }; + let k = &mp.key; + let r = &mp.rot; + let mul = &mp.mul; + let c = &mp.rc; + let mut s = String::with_capacity(6000); + s.push_str(&format!( + "// Memory-hard dataset core (MEMHARD.md). Cache: 2^{} words in 2^{} segments of {} chained ChaCha{} lines.\n", + cache_log2_words, + log2_segments(shape), + CACHE_LINES_PER_SEGMENT, + CHACHA_ROUNDS + )); + if mp.derive.is_some() { + s.push_str("// Item: 8 rounds of (the day's round program + one 64-byte cache read), then round program 8. All parameters are literals.\n"); + } else if m == 1 { + s.push_str("// Item: 8 rounds of seed-parameterised mixer + one 64-byte cache read, then a final mixer. All parameters are literals.\n"); + } else { + s.push_str(&format!("// Item: 8 rounds of {m} x seed-parameterised mixer + one 64-byte cache read, then {m} x final mixer (class v3, mixer multiplier {m},\n")); + s.push_str("// docs/plans/mixer-x4.md: the round key of application j of round r is 0x9E3779B9 * (r * m + j + 1)). All parameters are literals.\n"); + } + if let Some(dp) = &mp.derive { + s.push_str(&format!("// Counter ASIC 3.0 item 2 (docs/plans/counter-asic-3-derivation.md): the mixer slots run the day's drawn program, {} instructions per round\n", dp.len)); + s.push_str(&format!("// program (mh_round_0..{}), {} per item, drawn from the day key stream after the mixer constants (attempt {}, fingerprint {:016x}).\n", DERIVE_PROGRAMS - 1, dp.instr_count(), dp.attempt, dp.fingerprint())); + } + s.push_str(&format!("#define MH_CACHE_LINE_MASK {}\n", hex(cache_line_mask))); + s.push_str(&format!("#define MH_SEGMENT_LINES {}u\n", CACHE_LINES_PER_SEGMENT)); + s.push_str("#define MH_QR(a, b, c, d, r1, r2, r3, r4) { a += b; d ^= a; d = mh_rotl(d, r1); c += d; b ^= c; b = mh_rotl(b, r2); a += b; d ^= a; d = mh_rotl(d, r3); c += d; b ^= c; b = mh_rotl(b, r4); }\n"); + s.push_str(&format!( + "{fn_} {u} mh_rotl({u} x, {u} n) {{ return (x << n) | (x >> (32u - n)); }} // n in 1..31 at every call site\n" + )); + s.push('\n'); + s.push_str(&format!("// y = ChaCha{CHACHA_ROUNDS} core(x) + x\n")); + s.push_str(&format!("{fn_} void mh_chacha_block({lcptr} x, {lptr} y) {{\n")); + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) y[i] = x[i];\n")); + s.push_str(&format!(" for ({u} r = 0u; r < {}u; ++r) {{\n", CHACHA_ROUNDS / 2)); + s.push_str( + " MH_QR(y[0], y[4], y[8], y[12], 16u, 12u, 8u, 7u) MH_QR(y[1], y[5], y[9], y[13], 16u, 12u, 8u, 7u)\n", + ); + s.push_str( + " MH_QR(y[2], y[6], y[10], y[14], 16u, 12u, 8u, 7u) MH_QR(y[3], y[7], y[11], y[15], 16u, 12u, 8u, 7u)\n", + ); + s.push_str( + " MH_QR(y[0], y[5], y[10], y[15], 16u, 12u, 8u, 7u) MH_QR(y[1], y[6], y[11], y[12], 16u, 12u, 8u, 7u)\n", + ); + s.push_str( + " MH_QR(y[2], y[7], y[8], y[13], 16u, 12u, 8u, 7u) MH_QR(y[3], y[4], y[9], y[14], 16u, 12u, 8u, 7u)\n", + ); + s.push_str(" }\n"); + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) y[i] += x[i];\n")); + s.push_str("}\n"); + s.push('\n'); + s.push_str(&format!( + "// One cache segment: {} chained lines written at cache[seg * {}]. in_j = prev ^ (sigma || K || seg || j || tag), prev_0 = 0.\n", + CACHE_LINES_PER_SEGMENT, + CACHE_LINES_PER_SEGMENT * 16 + )); + s.push_str(&format!("{fn_} void mh_cache_segment({wptr} cache, {u} seg) {{\n")); + s.push_str(&format!(" {u} prev[16]; {u} x[16]; {u} y[16];\n")); + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) prev[i] = 0u;\n")); + s.push_str(&format!(" for ({u} j = 0u; j < MH_SEGMENT_LINES; ++j) {{\n")); + s.push_str(&format!( + " x[0] = {} ^ prev[0]; x[1] = {} ^ prev[1]; x[2] = {} ^ prev[2]; x[3] = {} ^ prev[3];\n", + hex(CHACHA_SIGMA[0]), + hex(CHACHA_SIGMA[1]), + hex(CHACHA_SIGMA[2]), + hex(CHACHA_SIGMA[3]) + )); + for i in 0..8 { + s.push_str(&format!(" x[{}] = {} ^ prev[{}];\n", 4 + i, hex(k[i]), 4 + i)); + } + s.push_str(&format!( + " x[12] = seg ^ prev[12]; x[13] = j ^ prev[13]; x[14] = {} ^ prev[14]; x[15] = {} ^ prev[15];\n", + hex(CACHE_TAG[0]), + hex(CACHE_TAG[1]) + )); + s.push_str(" mh_chacha_block(x, y);\n"); + s.push_str(&format!(" {wptr} line = cache + ((seg * MH_SEGMENT_LINES + j) * 16u);\n")); + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) {{ line[i] = y[i]; prev[i] = y[i]; }}\n")); + s.push_str(" }\n"); + s.push_str("}\n"); + s.push('\n'); + s.push_str("// M_r: per word (s ^ (RC + rk)) * MUL, then a column round and a diagonal round with the seed-drawn rotations.\n"); + s.push_str(&format!("{fn_} void mh_mixer({lptr} s, {u} rk) {{\n")); + for i in 0..16 { + s.push_str(&format!(" s[{i}] = (s[{i}] ^ ({} + rk)) * {};\n", hex(c[i]), hex(mul[i]))); + } + let col = (0..4).map(|i| format!("{}u", r[i])).collect::>().join(", "); + let dia = (4..8).map(|i| format!("{}u", r[i])).collect::>().join(", "); + s.push_str(&format!(" MH_QR(s[0], s[4], s[8], s[12], {col}) MH_QR(s[1], s[5], s[9], s[13], {col})\n")); + s.push_str(&format!(" MH_QR(s[2], s[6], s[10], s[14], {col}) MH_QR(s[3], s[7], s[11], s[15], {col})\n")); + s.push_str(&format!(" MH_QR(s[0], s[5], s[10], s[15], {dia}) MH_QR(s[1], s[6], s[11], s[12], {dia})\n")); + s.push_str(&format!(" MH_QR(s[2], s[7], s[8], s[13], {dia}) MH_QR(s[3], s[4], s[9], s[14], {dia})\n")); + s.push_str("}\n"); + s.push('\n'); + if m == 1 { + s.push_str(&format!( + "// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of mixer + cache line s[0] & mask; final mixer.\n" + )); + } else { + s.push_str(&format!( + "// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of {m} x mixer + cache line s[0] & mask; {m} x final mixer.\n" + )); + } + if let Some(dp) = &mp.derive { + // the nine round programs as straight-line functions over the 16-word state (the same text in every dialect) + for (r, prog) in dp.rounds.iter().enumerate() { + s.push_str(&format!("// Round program {r}: {} instructions, chain rule (every instruction reads the register the previous one wrote; s[0] first).\n", prog.len())); + s.push_str(&format!("{fn_} void mh_round_{r}({lptr} s) {{\n")); + for ins in prog { + s.push_str(" "); + s.push_str(&derive_instr_text(ins)); + s.push('\n'); + } + s.push_str("}\n"); + } + s.push_str(&format!("// Item t: 16 words. s = (K, t * MUL[i] + RC[i]); {ITEM_ROUNDS} rounds of (round program r, cache line s[0] & mask); round program {ITEM_ROUNDS}.\n")); + s.push_str(&item_signature(shape.state, fn_, cptr, u, lptr)); + for i in 0..8 { + s.push_str(&format!(" s[{i}] = {};\n", hex(k[i]))); + } + for i in 0..8 { + s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(mul[i]), hex(c[i]))); + } + if shape.state { + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= leaf[i];\n")); + } + for r in 0..ITEM_ROUNDS { + s.push_str(&format!(" mh_round_{r}(s);\n")); + s.push_str(&format!(" {{ {cptr} line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u); for ({u} i = 0u; i < 16u; ++i) s[i] ^= line[i]; }}\n")); + } + s.push_str(&format!(" mh_round_{ITEM_ROUNDS}(s);\n")); + s.push_str("}\n"); + return finish_memhard_core(s, layout, u, fn_, cptr, shape.state); + } + s.push_str(&item_signature(shape.state, fn_, cptr, u, lptr)); + for i in 0..8 { + s.push_str(&format!(" s[{i}] = {};\n", hex(k[i]))); + } + for i in 0..8 { + s.push_str(&format!(" s[{}] = t * {} + {};\n", 8 + i, hex(mul[i]), hex(c[i]))); + } + if shape.state { + // class v5: the window's state leaf of item t, before the first mixer (docs/design/class-v5-stored-state.md) + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= leaf[i];\n")); + } + s.push_str(&format!(" for ({u} r = 0u; r < {ITEM_ROUNDS}u; ++r) {{\n")); + if m == 1 { + s.push_str(" mh_mixer(s, 0x9E3779B9u * (r + 1u));\n"); + } else { + s.push_str(&format!(" for ({u} j = 0u; j < {m}u; ++j) mh_mixer(s, 0x9E3779B9u * (r * {m}u + j + 1u));\n")); + } + s.push_str(&format!(" {cptr} line = cache + ((s[0] & MH_CACHE_LINE_MASK) * 16u);\n")); + s.push_str(&format!(" for ({u} i = 0u; i < 16u; ++i) s[i] ^= line[i];\n")); + s.push_str(" }\n"); + if m == 1 { + s.push_str(&format!(" mh_mixer(s, 0x9E3779B9u * {}u);\n", ITEM_ROUNDS + 1)); + } else { + s.push_str(&format!( + " for ({u} j = 0u; j < {m}u; ++j) mh_mixer(s, 0x9E3779B9u * ({}u + j + 1u));\n", + ITEM_ROUNDS as u32 * m + )); + } + s.push_str("}\n"); + finish_memhard_core(s, layout, u, fn_, cptr, shape.state) +} + +/// The `mh_item` signature: under a state shape (class v5) the item takes its 16-word leaf (`leaves + 16 (t mod n)`). +fn item_signature(state: bool, fn_: &str, cptr: &str, u: &str, lptr: &str) -> String { + if state { + format!("{fn_} void mh_item({cptr} cache, {cptr} leaf, {u} t, {lptr} s) {{\n") + } else { + format!("{fn_} void mh_item({cptr} cache, {u} t, {lptr} s) {{\n") + } +} + +/// The tail of the memhard core: `mh_word` (and the era layout helpers) after `mh_item`. Under a state shape +/// `mh_word` takes the leaves and their count and derives item t's leaf as `leaves + 16 (t mod nLeaves)`. +fn finish_memhard_core(mut s: String, layout: Layout, u: &str, fn_: &str, cptr: &str, state: bool) -> String { + if state { + s.push_str("// Class v5 (docs/design/class-v5-stored-state.md): leaf(t) = leaves[t mod nLeaves], 16 words per leaf (leaves.bin).\n"); + s.push_str(&format!("{fn_} {cptr} mh_leaf({cptr} leaves, {u} nLeaves, {u} t) {{ return leaves + ((t % nLeaves) * 16u); }}\n")); + } + if layout.is_linear() { + s.push_str("// dataset[w] without the dataset: derive item w >> 4 and take word w & 15.\n"); + if state { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {cptr} leaves, {u} nLeaves, {u} w) {{ {u} s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, w >> 4u), w >> 4u, s); return s[w & 15u]; }}\n" + )); + } else { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, w >> 4u, s); return s[w & 15u]; }}\n" + )); + } + } else { + s.push_str(&layout_helpers(layout, u, fn_)); + s.push_str("// dataset[w] without the dataset: derive item mh_t(w) and take word mh_j(w).\n"); + if state { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {cptr} leaves, {u} nLeaves, {u} w) {{ {u} s[16]; mh_item(cache, mh_leaf(leaves, nLeaves, mh_t(w)), mh_t(w), s); return s[mh_j(w)]; }}\n" + )); + } else { + s.push_str(&format!( + "{fn_} {u} mh_word({cptr} cache, {u} w) {{ {u} s[16]; mh_item(cache, mh_t(w), s); return s[mh_j(w)]; }}\n" + )); + } + } + s +} + +/// The dataset store of one item in a build kernel: `d[i] = s[i]` at `ds + t * 16` for the linear layout, else the +/// scatter `ds[mh_addr(t, i)] = s[i]` (era layout). +fn build_store(layout: Layout, dialect: CoreDialect, ds: &str, t: &str) -> String { + let (u, wptr, cast) = match dialect { + CoreDialect::Metal => ("uint", "device uint*", ""), + CoreDialect::Cuda => ("uint32_t", "uint32_t*", "(size_t)"), + CoreDialect::OpenCl => ("uint", "__global uint*", "(ulong)"), + }; + if layout.is_linear() { + match dialect { + CoreDialect::Metal => format!(" {wptr} d = {ds} + {t} * 16u;\n for ({u} i = 0u; i < 16u; ++i) d[i] = s[i];\n"), + CoreDialect::Cuda => format!(" {wptr} d = {ds} + (size_t){t} * 16u;\n for ({u} i = 0u; i < 16u; ++i) d[i] = s[i];\n"), + CoreDialect::OpenCl => format!(" {wptr} d = {ds} + ((ulong){t} * 16u);\n for ({u} i = 0u; i < 16u; ++i) d[i] = s[i];\n"), + } + } else { + let indent = if dialect == CoreDialect::Metal { " " } else { " " }; + format!("{indent}for ({u} i = 0u; i < 16u; ++i) {ds}[{cast}mh_addr({t}, i)] = s[i];\n") + } +} + +/// [`metal_memhard`] plus, for a hot pack, the hot table's `ht_segment` and `igneum_hot_fill` kernel (one thread per +/// segment, `IGNEUM_HOT_SEGMENTS` threads). Byte-identical to [`metal_memhard`] for every other class. +pub fn metal_memhard_for(p: &Program, mp: &MixParams) -> String { + let mut s = metal_memhard_layout(mp, p.class.layout()); + if p.has_hot() { + s.push('\n'); + s.push_str(&emit_hot_core(p, CoreDialect::Metal)); + s.push_str("// Hot table: one thread per segment (IGNEUM_HOT_SEGMENTS threads).\n"); + s.push_str("kernel void igneum_hot_fill(device uint* hot [[buffer(0)]], uint gid [[thread_position_in_grid]]) {\n"); + s.push_str(" ht_segment(hot, gid);\n"); + s.push_str("}\n"); + } + s +} + +/// Metal library with the cache fill and dataset build kernels for one day key (`memhardMSL`, memhard.metal). +pub fn metal_memhard(mp: &MixParams) -> String { + metal_memhard_layout(mp, Layout::LINEAR) +} + +/// [`metal_memhard`] with the dataset layout (era layout). +pub fn metal_memhard_layout(mp: &MixParams, layout: Layout) -> String { + let mut s = String::new(); + s.push_str("#include \n"); + s.push_str("using namespace metal;\n"); + s.push_str(&emit_memhard_core_layout(mp, CoreDialect::Metal, layout)); + s.push('\n'); + s.push_str(&format!("// One thread per segment (2^{} threads).\n", log2_segments(&mp.shape))); + s.push_str( + "kernel void igneum_cache_fill(device uint* cache [[buffer(0)]], uint gid [[thread_position_in_grid]]) {\n", + ); + s.push_str(" mh_cache_segment(cache, gid);\n"); + s.push_str("}\n"); + s.push_str("// One thread per 64-byte item (dataset words / 16 threads).\n"); + if mp.shape.state { + s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) in buffer 2, their count in buffer 3.\n"); + s.push_str( + "kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n", + ); + s.push_str(" device const uint* leaves [[buffer(2)]], constant uint& nLeaves [[buffer(3)]],\n"); + s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); + s.push_str(" uint s[16];\n"); + s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, gid), gid, s);\n"); + } else { + s.push_str( + "kernel void igneum_build(device const uint* cache [[buffer(0)]], device uint* dataset [[buffer(1)]],\n", + ); + s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); + s.push_str(" uint s[16];\n"); + s.push_str(" mh_item(cache, gid, s);\n"); + } + s.push_str(&build_store(layout, CoreDialect::Metal, "dataset", "gid")); + s.push_str("}\n"); + s +} + +const DS_ELEM_BODY: &str = " x *= 0x9E3779B1u; x ^= x >> 15;\n x += d1;\n x *= 0x85EBCA77u; x ^= x >> 13;\n x *= 0xC2B2AE3Du; x ^= x >> 16;\n return x;\n}\n"; + +/// The Metal hash kernel (`generateMSL`, program.metal). +pub fn metal_program(p: &Program, dataset_log2: u32, source: LoadSource) -> String { + metal_program_impl(p, dataset_log2, source, false) +} + +/// The header-bound Metal kernel (`program_bound.metal`, serve mode of proto-metal): `igneum_hash_bound` reads its +/// init words `I` from `constant uint* initw [[buffer(3)]]` (`bind::block_init_words`) instead of `SEEDW`. Same +/// instruction text as `igneum_hash`. Stored dataset only. +pub fn metal_program_bound(p: &Program, dataset_log2: u32) -> String { + metal_program_impl(p, dataset_log2, LoadSource::Stored, true) +} + +fn metal_program_impl(p: &Program, dataset_log2: u32, source: LoadSource, bound: bool) -> String { + let mask = mask_for(dataset_log2); + let mut s = String::with_capacity(5000); + s.push_str("#include \n"); + s.push_str("using namespace metal;\n"); + s.push('\n'); + s.push_str(&format!("#define MASK {}\n", hex(mask))); + s.push_str(&hot_define(p)); + s.push_str(&format!("constant uint SEEDW[8] = {{ {} }};\n", join_hex(&p.seed))); + s.push('\n'); + s.push_str("inline uint splitmix32(uint x) {\n"); + s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); + s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n"); + s.push_str(" x ^= x >> 16;\n"); + s.push_str(" return x;\n"); + s.push_str("}\n"); + s.push_str("inline uint rotl_imm(uint x, uint n) { return (x << n) | (x >> (32u - n)); } // n in 1..31\n"); + s.push_str("inline uint rotr_var(uint x, uint n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n"); + s.push_str("inline uint ds_elem(uint i, uint d0, uint d1) {\n"); + s.push_str(" uint x = i ^ d0;\n"); + s.push_str(DS_ELEM_BODY); + s.push('\n'); + if p.has_wide() { + s.push_str("#define WMASK (MASK & ~31u)\n\n"); + } + let mut buffer0 = "device const uint* dataset [[buffer(0)]]"; + if let LoadSource::InlineMemhard(mp) = &source { + s.push_str(&emit_memhard_core_layout(mp, CoreDialect::Metal, p.class.layout())); + s.push('\n'); + buffer0 = "device const uint* cache [[buffer(0)]]"; + } + s.push_str(&scratch_prelude(p, CoreDialect::Metal)); + if bound { + s.push_str("// Header-bound variant: the init words come from buffer 3 (bind.rs), not from SEEDW.\n"); + s.push_str(&format!("kernel void igneum_hash_bound({buffer0},\n")); + } else { + s.push_str(&format!("kernel void igneum_hash({buffer0},\n")); + } + s.push_str(" device ulong* out [[buffer(1)]],\n"); + s.push_str(" constant uint& baseNonce [[buffer(2)]],\n"); + if bound { + s.push_str(" constant uint* initw [[buffer(3)]],\n"); + } + if p.has_hot() { + s.push_str(&format!(" device const uint* hot [[buffer({})]],\n", if bound { 4 } else { 3 })); + } + if p.has_scratch() { + let b = (if bound { 4 } else { 3 }) + p.has_hot() as usize; + s.push_str(&format!(" device uint* scratch [[buffer({b})]],\n")); + s.push_str(&format!(" constant uint& groups [[buffer({})]],\n", b + 1)); + s.push_str(&format!(" constant uint& salt [[buffer({})]],\n", b + 2)); + s.push_str(" uint tid [[thread_position_in_grid]],\n"); + s.push_str(" uint nthreads [[threads_per_grid]]) {\n"); + s.push_str(&persistent_prologue(CoreDialect::Metal, p.class.scratch_words_per_lane())); + } else { + s.push_str(" uint gid [[thread_position_in_grid]]) {\n"); + } + s.push_str(" uint nonce = baseNonce + gid;\n"); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + if p.has_wide() { + s.push_str(" uint lane = gid & 31u;\n"); + } + let iw = if bound { "initw" } else { "SEEDW" }; + for i in 0..8 { + s.push_str(&format!( + " {{ uint x = nonce ^ {iw}[{i}]; x += 0x9e3779b9u * {}u; x = splitmix32(x); r{i} = x ^ {iw}[{}]; }}\n", + i + 1, + (i + 1) & 7 + )); + } + s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); + let era = p.class.era; + let word_index = |a: &str, wide: bool, ins: &Instr| -> String { + if wide { + format!("(simd_broadcast({a}, 0) & WMASK) + lane") + } else { + load_index_expr(CoreDialect::Metal, era.as_ref(), ins, a, dataset_log2) + } + }; + let fetch = |idx: String| -> String { + match &source { + LoadSource::Stored => format!("dataset[{idx}]"), + LoadSource::InlineClosed(d0, d1) => format!("ds_elem({idx}, {}, {})", hex(*d0), hex(*d1)), + LoadSource::InlineMemhard(_) => format!("mh_word(cache, {idx})"), + } + }; + for (k, ins) in p.instrs.iter().enumerate() { + let d = format!("r{}", ins.dst); + let a = format!("r{}", ins.src); + let b = format!("r{}", ins.src2); + let line = match ins.op { + Op::Add => format!( + "{d} = {d} + {a} + select({}, {}, ((sel >> {}u) & 1u) != 0u);", + hex(ins.imm), + hex(ins.imm2), + ins.bit + ), + Op::Sub => format!("{d} = {d} - {a};"), + Op::Mul => format!("{d} = {d} * {a};"), + Op::MulHi => format!("{d} = mulhi({d}, {a});"), + Op::Xor => format!("{d} = {d} ^ {a};"), + Op::Or => format!("{d} = {d} | {a};"), + Op::Rotl => format!("{d} = rotl_imm({d}, {}u);", ins.rot), + Op::Rotr => format!("{d} = rotr_var({d}, {a});"), + Op::Mad => format!("{d} = {a} * {b} + {d};"), + Op::Shfl => format!("{d} = {d} ^ simd_shuffle_xor({a}, (ushort){});", ins.mask), + Op::Load if load_width(ins) > 1 => { + let (src, closed) = match &source { + LoadSource::Stored => (WideSource::Stored, None), + LoadSource::InlineClosed(d0, d1) => (WideSource::InlineClosed, Some((*d0, *d1))), + LoadSource::InlineMemhard(_) => (WideSource::InlineMemhard, None), + }; + wide_load_stmt(CoreDialect::Metal, &d, &word_index(&a, false, ins), ins.width, src, closed) + } + Op::Load => format!("{d} = {d} ^ {};", fetch(word_index(&a, false, ins))), + Op::WLoad => format!("{d} = {d} ^ {};", fetch(word_index(&a, true, ins))), + Op::Scratch => scratch_stmt(CoreDialect::Metal, &d, &a, p.class.scratch_slot_mask()), + Op::Hot => hot_stmt(CoreDialect::Metal, &d, &a), + }; + s.push_str(&format!(" {line} // {k}\n")); + } + s.push_str(&shadow_block(p, CoreDialect::Metal)); + s.push_str(" }\n"); + s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); + s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); + s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); + if p.has_scratch() { + s.push_str(" }\n"); + } + s.push_str("}\n"); + s +} + +/// The Metal closed-form fill kernel (`fillMSL`). +pub const METAL_FILL: &str = "#include \nusing namespace metal;\ninline uint ds_elem(uint i, uint d0, uint d1) {\n uint x = i ^ d0;\n x *= 0x9E3779B1u; x ^= x >> 15;\n x += d1;\n x *= 0x85EBCA77u; x ^= x >> 13;\n x *= 0xC2B2AE3Du; x ^= x >> 16;\n return x;\n}\nkernel void igneum_fill(device uint* dataset [[buffer(0)]],\n constant uint2& day [[buffer(1)]],\n uint gid [[thread_position_in_grid]]) {\n dataset[gid] = ds_elem(gid, day.x, day.y);\n}"; + +fn generated_by(seed: &str) -> String { + format!("// Generated by igneum-pow export (generator v{GENERATOR_VERSION}) for seed \"{seed}\". Do not edit by hand.\n") +} + +pub fn hex_bytes(b: &[u8]) -> String { + b.iter().map(|x| format!("{x:02x}")).collect() +} + +fn init_line(p: &Program, u: &str, i: usize) -> String { + let addc = 0x9e3779b9u32.wrapping_mul(i as u32 + 1); + format!( + " {{ {u} x = nonce ^ {}; x += {}; x = splitmix32(x); r{i} = x ^ {}; }} // SEEDW[{i}], 0x9e3779b9u * {}u, SEEDW[{}]\n", + hex(p.seed[i]), + hex(addc), + hex(p.seed[(i + 1) & 7]), + i + 1, + (i + 1) & 7 + ) +} + +/// The instruction lines of the CUDA hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`). +fn cuda_instr_lines(p: &Program, dataset_log2: u32) -> String { + let mut s = String::with_capacity(6000); + let era = p.class.era; + for (k, ins) in p.instrs.iter().enumerate() { + let d = format!("r{}", ins.dst); + let a = format!("r{}", ins.src); + let b = format!("r{}", ins.src2); + let line = match ins.op { + // Metal select(A, B, c) returns c ? B : A, so the true branch is imm2 here as well. + Op::Add => format!( + "{d} = {d} + {a} + ((((sel >> {}u) & 1u) != 0u) ? {} : {});", + ins.bit, + hex(ins.imm2), + hex(ins.imm) + ), + Op::Sub => format!("{d} = {d} - {a};"), + Op::Mul => format!("{d} = {d} * {a};"), + Op::MulHi => format!("{d} = __umulhi({d}, {a});"), + Op::Xor => format!("{d} = {d} ^ {a};"), + Op::Or => format!("{d} = {d} | {a};"), + Op::Rotl => format!("{d} = rotl_imm({d}, {}u);", ins.rot), + Op::Rotr => format!("{d} = rotr_var({d}, {a});"), + Op::Mad => format!("{d} = {a} * {b} + {d};"), + Op::Shfl => format!("{d} = {d} ^ __shfl_xor_sync(0xffffffffu, {a}, {});", ins.mask), + Op::Load if load_width(ins) > 1 => { + wide_load_stmt(CoreDialect::Cuda, &d, &load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None) + } + Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::Cuda, era.as_ref(), ins, &a, dataset_log2)), + Op::WLoad => format!("{d} = {d} ^ ds[(__shfl_sync(0xffffffffu, {a}, 0) & wmask) + lane];"), + Op::Scratch => scratch_stmt(CoreDialect::Cuda, &d, &a, p.class.scratch_slot_mask()), + Op::Hot => hot_stmt(CoreDialect::Cuda, &d, &a), + }; + s.push_str(&format!(" {line} // {k} {}\n", ins.op.name())); + } + s +} + +/// The CUDA kernel (`generateCUDA`, kernel.cu). `memhard` is `None` for a closed-form pack. +pub fn cuda_kernel(p: &Program, memhard: Option<&MixParams>) -> String { + cuda_kernel_at(p, memhard, DEFAULT_DATASET_LOG2) +} + +/// [`cuda_kernel`] at a dataset size (an era program's window constants are literals of the pack's size; every +/// other class ignores it). +pub fn cuda_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { + let layout = p.class.layout(); + let mut s = String::with_capacity(9000); + s.push_str(&generated_by(&p.seed_string)); + s.push_str( + "// Bit-exact twin of the Metal kernel for the same seed (see proto-cuda/CHECKLIST.md and program.metal).\n", + ); + s.push_str("// Compiled ahead of time by nvcc together with proto-cuda/host.cu. No NVRTC.\n"); + s.push_str("#include \n"); + s.push_str("#include \n"); + s.push_str("#include \"program.h\"\n"); + if memhard.is_some() { + s.push_str("#include \"memhard.h\"\n"); + } + s.push('\n'); + s.push_str(&hot_define(p)); + s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n"); + s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); + s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n"); + s.push_str(" x ^= x >> 16;\n"); + s.push_str(" return x;\n"); + s.push_str("}\n"); + s.push_str("// n is a literal in 1..31 at every call site, so both shift amounts are in 1..31.\n"); + s.push_str("__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }\n"); + s.push_str("// n is masked to 0..31; the second shift amount is masked too, so n == 0 gives x.\n"); + s.push_str("__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n"); + s.push_str("__device__ __forceinline__ uint32_t ds_elem(uint32_t i, uint32_t d0, uint32_t d1) {\n"); + s.push_str(" uint32_t x = i ^ d0;\n"); + s.push_str(DS_ELEM_BODY); + s.push('\n'); + if memhard.is_none() { + s.push_str("// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.\n"); + s.push_str("__global__ void igneum_fill(uint32_t* ds, uint32_t n, uint32_t d0, uint32_t d1) {\n"); + s.push_str(" uint32_t i = blockIdx.x * blockDim.x + threadIdx.x;\n"); + s.push_str(" if (i < n) ds[i] = ds_elem(i, d0, d1);\n"); + s.push_str("}\n"); + s.push('\n'); + } else { + s.push_str( + "// Memory-hard dataset (MEMHARD.md). One thread per cache segment; one thread per 64-byte dataset item.\n", + ); + s.push_str( + "// The core functions (mh_cache_segment, mh_item) are in memhard.h and are also compiled for the host.\n", + ); + s.push_str("__global__ void igneum_cache_fill(uint32_t* cache, uint32_t nSegments) {\n"); + s.push_str(" uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;\n"); + s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n"); + s.push_str("}\n"); + if p.class.state { + s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.\n"); + s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {\n"); + } else { + s.push_str("__global__ void igneum_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n"); + } + s.push_str(" uint32_t t = blockIdx.x * blockDim.x + threadIdx.x;\n"); + s.push_str(" if (t < nItems) {\n"); + s.push_str(" uint32_t s[16];\n"); + if p.class.state { + s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);\n"); + } else { + s.push_str(" mh_item(cache, t, s);\n"); + } + s.push_str(&build_store(layout, CoreDialect::Cuda, "ds", "t")); + s.push_str(" }\n"); + s.push_str("}\n"); + if p.has_hot() { + s.push_str("// Hot table (ht_segment is in memhard.h): one thread per segment.\n"); + s.push_str("__global__ void igneum_hot_fill(uint32_t* hot, uint32_t nSegments) {\n"); + s.push_str(" uint32_t seg = blockIdx.x * blockDim.x + threadIdx.x;\n"); + s.push_str(" if (seg < nSegments) ht_segment(hot, seg);\n"); + s.push_str("}\n"); + } + s.push('\n'); + } + if memhard.is_none() && p.has_hot() { + panic!("a hot-table pack needs the memory-hard dataset (the hot fill shares its ChaCha core)"); + } + s.push_str("// One hash per thread. blockDim.x is a multiple of 32; lane = threadIdx.x & 31 and every\n"); + s.push_str("// __shfl_xor_sync stays inside the lane's own warp, exactly like simd_shuffle_xor inside a\n"); + s.push_str("// 32-wide Metal SIMD group. Control flow is uniform, so the full 0xffffffff member mask is valid.\n"); + s.push_str(&scratch_prelude(p, CoreDialect::Cuda)); + let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" }; + let hot_args = if p.has_hot() { ", const uint32_t* hot" } else { "" }; + s.push_str(&format!("__global__ void igneum_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask{hot_args}{scratch_args}) {{\n")); + if p.has_scratch() { + s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane())); + } else { + s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); + } + s.push_str(" uint32_t nonce = baseNonce + gid;\n"); + s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n"); + if p.has_wide() { + s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n"); + } + for i in 0..8 { + s.push_str(&init_line(p, "uint32_t", i)); + } + s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n")); + s.push_str(&cuda_instr_lines(p, dataset_log2)); + s.push_str(&shadow_block(p, CoreDialect::Cuda)); + s.push_str(" }\n"); + s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); + s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); + s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n"); + if p.has_scratch() { + s.push_str(" }\n"); + } + s.push_str("}\n"); + s.push('\n'); + s.push_str("// Host-side launch wrappers. Declared in program.h, called from host.cu.\n"); + if memhard.is_none() { + s.push_str("cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1) {\n"); + s.push_str(" if (nWords == 0u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 256u;\n"); + s.push_str(" uint32_t grid = (nWords + block - 1u) / block;\n"); + s.push_str(" igneum_fill<<>>(ds, nWords, d0, d1);\n"); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + s.push('\n'); + } else { + s.push_str("cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments) {\n"); + s.push_str(" if (nSegments == 0u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 256u;\n"); + s.push_str(" uint32_t grid = (nSegments + block - 1u) / block;\n"); + s.push_str(" igneum_cache_fill<<>>(cache, nSegments);\n"); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + s.push('\n'); + if p.class.state { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems) {\n"); + s.push_str(" if (nItems == 0u || nLeaves == 0u) return cudaErrorInvalidValue;\n"); + } else { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems) {\n"); + s.push_str(" if (nItems == 0u) return cudaErrorInvalidValue;\n"); + } + s.push_str(" uint32_t block = 256u;\n"); + s.push_str(" uint32_t grid = (nItems + block - 1u) / block;\n"); + if p.class.state { + s.push_str(" igneum_build<<>>(ds, cache, leaves, nLeaves, nItems);\n"); + } else { + s.push_str(" igneum_build<<>>(ds, cache, nItems);\n"); + } + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + s.push('\n'); + if p.has_hot() { + s.push_str("cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments) {\n"); + s.push_str(" if (nSegments == 0u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 256u;\n"); + s.push_str(" uint32_t grid = (nSegments + block - 1u) / block;\n"); + s.push_str(" igneum_hot_fill<<>>(hot, nSegments);\n"); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + s.push('\n'); + } + } + let (hot_decl, hot_pass) = if p.has_hot() { (" const uint32_t* hot,", " hot,") } else { ("", "") }; + if p.has_scratch() { + s.push_str("// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).\n"); + s.push_str(&format!("cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n")); + s.push_str(" uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {\n"); + s.push_str(" if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 32u * blockWarps;\n"); + s.push_str(" if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(&format!(" igneum_hash<<>>(ds, out, baseNonce, mask,{hot_pass} scratch, nonces / 32u, salt);\n")); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + } else { + s.push_str(&format!( + "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n", + )); + s.push_str(" uint32_t nonces, uint32_t blockWarps) {\n"); + s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 32u * blockWarps;\n"); + s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(&format!(" igneum_hash<<>>(ds, out, baseNonce, mask{});\n", if p.has_hot() { ", hot" } else { "" })); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + } + s.push('\n'); + s.push_str("cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {\n"); + s.push_str(" cudaFuncAttributes attr;\n"); + s.push_str(" cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash);\n"); + s.push_str(" if (e != cudaSuccess) return e;\n"); + s.push_str(" *numRegs = attr.numRegs;\n"); + s.push_str(" return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash, (int)(32u * blockWarps), 0);\n"); + s.push_str("}\n"); + s +} + +/// `kernel_bound.cu`: the header-bound CUDA hash kernel for the serve mode of proto-cuda. A standalone +/// translation unit (compiled next to kernel.cu, which keeps the cache-fill and build wrappers): the init words +/// `I` arrive by value in `IgneumInitWords` (`bind::block_init_words`), the instruction text is that of +/// `igneum_hash`. Declarations for the host are at the top of the file. +pub fn cuda_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { + cuda_kernel_bound_at(p, memhard, DEFAULT_DATASET_LOG2) +} + +/// [`cuda_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]). +pub fn cuda_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { + let mut s = String::with_capacity(9000); + s.push_str(&generated_by(&p.seed_string)); + s.push_str( + "// Header-bound twin of igneum_hash in kernel.cu: the init words come from a kernel argument, not SEEDW.\n", + ); + s.push_str("// Host declarations (also in program_bound.h if present):\n"); + s.push_str("// struct IgneumInitWords { uint32_t w[8]; };\n"); + s.push_str("// cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n"); + s.push_str( + "// IgneumInitWords iw, uint32_t nonces, uint32_t blockWarps);\n", + ); + s.push_str("// cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);\n"); + s.push_str("#include \n"); + s.push_str("#include \n"); + s.push_str("#include \"program.h\"\n"); + s.push('\n'); + s.push_str("struct IgneumInitWords { uint32_t w[8]; };\n"); + s.push('\n'); + s.push_str(&hot_define(p)); + s.push_str("__device__ __forceinline__ uint32_t splitmix32(uint32_t x) {\n"); + s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); + s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n"); + s.push_str(" x ^= x >> 16;\n"); + s.push_str(" return x;\n"); + s.push_str("}\n"); + s.push_str("__device__ __forceinline__ uint32_t rotl_imm(uint32_t x, uint32_t n) { return (x << n) | (x >> (32u - n)); }\n"); + s.push_str("__device__ __forceinline__ uint32_t rotr_var(uint32_t x, uint32_t n) { n &= 31u; return (x >> n) | (x << ((32u - n) & 31u)); }\n"); + s.push('\n'); + let _ = memhard; // the bound kernel reads the stored dataset in both constructions + s.push_str(&scratch_prelude(p, CoreDialect::Cuda)); + let scratch_args = if p.has_scratch() { ", uint32_t* scratch, uint32_t groups, uint32_t salt" } else { "" }; + let hot_args = if p.has_hot() { ", const uint32_t* hot" } else { "" }; + s.push_str(&format!("__global__ void igneum_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask, IgneumInitWords iw{hot_args}{scratch_args}) {{\n")); + if p.has_scratch() { + s.push_str(&persistent_prologue(CoreDialect::Cuda, p.class.scratch_words_per_lane())); + } else { + s.push_str(" uint32_t gid = blockIdx.x * blockDim.x + threadIdx.x;\n"); + } + s.push_str(" uint32_t nonce = baseNonce + gid;\n"); + s.push_str(" uint32_t r0, r1, r2, r3, r4, r5, r6, r7;\n"); + if p.has_wide() { + s.push_str(" uint32_t lane = threadIdx.x & 31u;\n uint32_t wmask = mask & ~31u;\n"); + } + for i in 0..8 { + s.push_str(&format!( + " {{ uint32_t x = nonce ^ iw.w[{i}]; x += 0x9e3779b9u * {}u; x = splitmix32(x); r{i} = x ^ iw.w[{}]; }}\n", + i + 1, + (i + 1) & 7 + )); + } + s.push_str(&format!("\n for (uint32_t it = 0u; it < {ITERATIONS}u; ++it) {{\n uint32_t sel = r0;\n")); + s.push_str(&cuda_instr_lines(p, dataset_log2)); + s.push_str(&shadow_block(p, CoreDialect::Cuda)); + s.push_str(" }\n"); + s.push_str(" uint32_t lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); + s.push_str(" uint32_t hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); + s.push_str(" out[gid] = ((uint64_t)hi << 32) | (uint64_t)lo;\n"); + if p.has_scratch() { + s.push_str(" }\n"); + } + s.push_str("}\n"); + s.push('\n'); + let (hot_decl, hot_pass) = if p.has_hot() { (" const uint32_t* hot,", ", hot") } else { ("", "") }; + if p.has_scratch() { + s.push_str("// Variant 5: the wrapper launches `warps` persistent warps over `nonces / 32` units (host.cu does not use it).\n"); + s.push_str("cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n"); + s.push_str(&format!(" IgneumInitWords iw,{hot_decl} uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt) {{\n")); + s.push_str(" if (blockWarps == 0u || blockWarps > 32u || warps == 0u || (warps % blockWarps) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 32u * blockWarps;\n"); + s.push_str(" if (nonces == 0u || (nonces % (32u * warps)) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(&format!(" igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw{hot_pass}, scratch, nonces / 32u, salt);\n")); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + } else { + s.push_str( + "cudaError_t igneum_launch_hash_bound(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,\n", + ); + s.push_str(&format!(" IgneumInitWords iw,{hot_decl} uint32_t nonces, uint32_t blockWarps) {{\n")); + s.push_str(" if (blockWarps == 0u || blockWarps > 32u) return cudaErrorInvalidValue;\n"); + s.push_str(" uint32_t block = 32u * blockWarps;\n"); + s.push_str(" if (nonces == 0u || (nonces % block) != 0u) return cudaErrorInvalidValue;\n"); + s.push_str(&format!(" igneum_hash_bound<<>>(ds, out, baseNonce, mask, iw{hot_pass});\n")); + s.push_str(" return cudaGetLastError();\n"); + s.push_str("}\n"); + } + s.push('\n'); + s.push_str("cudaError_t igneum_hash_bound_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps) {\n"); + s.push_str(" cudaFuncAttributes attr;\n"); + s.push_str(" cudaError_t e = cudaFuncGetAttributes(&attr, igneum_hash_bound);\n"); + s.push_str(" if (e != cudaSuccess) return e;\n"); + s.push_str(" *numRegs = attr.numRegs;\n"); + s.push_str(" return cudaOccupancyMaxActiveBlocksPerMultiprocessor(blocksPerSM, igneum_hash_bound, (int)(32u * blockWarps), 0);\n"); + s.push_str("}\n"); + s +} + +/// The instruction lines of the OpenCL hash kernel body (shared by `igneum_hash` and `igneum_hash_bound`). +fn opencl_instr_lines(p: &Program, dataset_log2: u32) -> String { + let mut s = String::with_capacity(6000); + let era = p.class.era; + for (k, ins) in p.instrs.iter().enumerate() { + let d = format!("r{}", ins.dst); + let a = format!("r{}", ins.src); + let b = format!("r{}", ins.src2); + let line = match ins.op { + Op::Add => format!( + "{d} = {d} + {a} + ((((sel >> {}u) & 1u) != 0u) ? {} : {});", + ins.bit, + hex(ins.imm2), + hex(ins.imm) + ), + Op::Sub => format!("{d} = {d} - {a};"), + Op::Mul => format!("{d} = {d} * {a};"), + Op::MulHi => format!("{d} = mul_hi({d}, {a});"), + Op::Xor => format!("{d} = {d} ^ {a};"), + Op::Or => format!("{d} = {d} | {a};"), + Op::Rotl => format!("{d} = rotl_imm({d}, {}u);", ins.rot), + Op::Rotr => format!("{d} = rotr_var({d}, {a});"), + Op::Mad => format!("{d} = {a} * {b} + {d};"), + Op::Shfl => format!("{{ uint t_; IGNEUM_SHFL_XOR(t_, {a}, {}u); {d} = {d} ^ t_; }}", ins.mask), + Op::Load if load_width(ins) > 1 => { + wide_load_stmt(CoreDialect::OpenCl, &d, &load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2), ins.width, WideSource::Stored, None) + } + Op::Load => format!("{d} = {d} ^ ds[{}];", load_index_expr(CoreDialect::OpenCl, era.as_ref(), ins, &a, dataset_log2)), + Op::WLoad => format!("{{ uint t_; IGNEUM_BCAST0(t_, {a}); {d} = {d} ^ ds[(t_ & wmask) + lane]; }}"), + Op::Scratch => scratch_stmt(CoreDialect::OpenCl, &d, &a, p.class.scratch_slot_mask()), + Op::Hot => hot_stmt(CoreDialect::OpenCl, &d, &a), + }; + s.push_str(&format!(" {line} // {k} {}\n", ins.op.name())); + } + s +} + +/// `kernel_bound.cl`: `kernel.cl` plus the header-bound kernel `igneum_hash_bound`, whose init words come from a +/// fifth argument (`__global const uint* initw`, 8 words, `bind::block_init_words`). One source file so the serve +/// mode of proto-opencl/host.c builds cache fill, dataset build and the bound hash from it at runtime. +pub fn opencl_kernel_bound(p: &Program, memhard: Option<&MixParams>) -> String { + opencl_kernel_bound_at(p, memhard, DEFAULT_DATASET_LOG2) +} + +/// [`opencl_kernel_bound`] at a dataset size (see [`cuda_kernel_at`]). +pub fn opencl_kernel_bound_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { + let mut s = opencl_kernel_at(p, memhard, dataset_log2); + s.push('\n'); + s.push_str( + "// Header-bound variant (bind.rs): the init words come from initw, not SEEDW. Same body as igneum_hash.\n", + ); + let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" }; + let hot_args = if p.has_hot() { ", __global const uint* hot" } else { "" }; + s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash_bound(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask, __global const uint* initw{hot_args}{scratch_args}) {{\n")); + let (setup, unit_loop) = persistent_prologue_parts(CoreDialect::OpenCl, p.class.scratch_words_per_lane()); + if p.has_scratch() { + s.push_str(&setup); + } else { + s.push_str(" uint gid = (uint)get_global_id(0);\n"); + } + s.push_str(" uint lid = (uint)get_local_id(0);\n"); + if p.has_scratch() { + // the __local exchange buffer must sit at the kernel's outermost scope: declare it, then open the unit loop + s.push_str("#if IGNEUM_EXCHANGE == 0\n"); + s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); + s.push_str(" uint xk = 0u;\n"); + s.push_str("#else\n"); + s.push_str(" (void)lid;\n"); + s.push_str("#endif\n"); + s.push_str(&unit_loop); + s.push_str(" uint nonce = baseNonce + gid;\n"); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n"); + } else { + s.push_str(" uint nonce = baseNonce + gid;\n"); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str(" uint iw0 = initw[0], iw1 = initw[1], iw2 = initw[2], iw3 = initw[3], iw4 = initw[4], iw5 = initw[5], iw6 = initw[6], iw7 = initw[7];\n"); + s.push_str("#if IGNEUM_EXCHANGE == 0\n"); + s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); + s.push_str(" uint xk = 0u;\n"); + s.push_str("#else\n"); + s.push_str(" (void)lid;\n"); + s.push_str("#endif\n"); + } + if p.has_wide() { + s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n"); + } + for i in 0..8 { + s.push_str(&format!( + " {{ uint x = nonce ^ iw{i}; x += 0x9e3779b9u * {}u; x = splitmix32(x); r{i} = x ^ iw{}; }}\n", + i + 1, + (i + 1) & 7 + )); + } + s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); + s.push_str(&opencl_instr_lines(p, dataset_log2)); + s.push_str(&shadow_block(p, CoreDialect::OpenCl)); + s.push_str(" }\n"); + s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); + s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); + s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); + if p.has_scratch() { + s.push_str(" }\n"); + } + s.push_str("}\n"); + s +} + +/// The OpenCL C 1.2 kernel (`generateOpenCL`, kernel.cl). +pub fn opencl_kernel(p: &Program, memhard: Option<&MixParams>) -> String { + opencl_kernel_at(p, memhard, DEFAULT_DATASET_LOG2) +} + +/// [`opencl_kernel`] at a dataset size (see [`cuda_kernel_at`]). +pub fn opencl_kernel_at(p: &Program, memhard: Option<&MixParams>, dataset_log2: u32) -> String { + let layout = p.class.layout(); + let mut s = String::with_capacity(14000); + s.push_str(&generated_by(&p.seed_string)); + s.push_str("// OpenCL C twin of the Metal kernel for the same seed (see proto-opencl/README.md, WAVEFRONT.md and program.metal).\n"); + s.push_str("// Built from source at runtime by proto-opencl/host.c, which passes these defines:\n"); + s.push_str("// IGNEUM_GROUP work-group size of igneum_hash, a multiple of 32 (default 32: one work-group = one 32-lane unit)\n"); + s.push_str( + "// IGNEUM_EXCHANGE 0 = local-memory exchange with a barrier (any device, any wave width; the default)\n", + ); + s.push_str("// 1 = sub_group_shuffle_xor (cl_khr_subgroup_shuffle), only with IGNEUM_GROUP 32 and a sub-group size of exactly 32\n"); + s.push_str("// 2 = intel_sub_group_shuffle_xor (cl_intel_subgroups), same condition\n"); + s.push_str("// The verification unit is always 32 lanes. A 64-wide hardware wave (AMD GCN/CDNA, RDNA in wave64) runs two units;\n"); + s.push_str("// the exchange masks are 1, 2, 4, 8, 16, so every partner lane lies inside the lane's own aligned run of 32.\n"); + s.push_str("#ifndef IGNEUM_GROUP\n#define IGNEUM_GROUP 32\n#endif\n"); + s.push_str("#ifndef IGNEUM_EXCHANGE\n#define IGNEUM_EXCHANGE 0\n#endif\n"); + s.push_str("#ifdef __OPENCL_VERSION__\n"); + s.push_str("#define IGNEUM_KERNEL_HASH __kernel __attribute__((reqd_work_group_size(IGNEUM_GROUP, 1, 1)))\n"); + s.push_str("#define IGNEUM_LOCAL_WORDS(name, n) __local uint name[n]\n"); + s.push_str("#if IGNEUM_EXCHANGE == 1\n"); + s.push_str("#ifdef cl_khr_subgroups\n#pragma OPENCL EXTENSION cl_khr_subgroups : enable\n#endif\n"); + s.push_str("#ifdef cl_khr_subgroup_shuffle\n#pragma OPENCL EXTENSION cl_khr_subgroup_shuffle : enable\n#endif\n"); + s.push_str("#elif IGNEUM_EXCHANGE == 2\n#pragma OPENCL EXTENSION cl_intel_subgroups : enable\n#endif\n"); + if p.has_scratch() { + s.push_str("#define IGNEUM_U4(a, b, c, d) ((uint4)((a), (b), (c), (d)))\n"); + } + s.push_str("#else\n"); + s.push_str("// Not an OpenCL compiler: proto-opencl/emu compiles this file as C++ and supplies the built-ins and these two macros.\n"); + s.push_str("#include \"emu_opencl.h\"\n#endif\n"); + s.push('\n'); + s.push_str("#if IGNEUM_EXCHANGE == 1\n"); + s.push_str("#define IGNEUM_SHFL_XOR(dst, a, m) dst = sub_group_shuffle_xor((a), (uint)(m))\n"); + s.push_str("#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)\n"); + s.push_str("#elif IGNEUM_EXCHANGE == 2\n"); + s.push_str("#define IGNEUM_SHFL_XOR(dst, a, m) dst = intel_sub_group_shuffle_xor((a), (uint)(m))\n"); + s.push_str("#define IGNEUM_BCAST0(dst, a) dst = sub_group_broadcast((a), 0u)\n"); + s.push_str("#else\n"); + s.push_str("// Local-memory exchange. Two buffers of IGNEUM_GROUP words alternate (xk counts exchanges), so one barrier per\n"); + s.push_str("// exchange is enough: a lane can only overwrite buffer b at exchange k+2 after passing barrier k+1, and every lane\n"); + s.push_str("// reaches barrier k+1 only after its read of buffer b at exchange k. The partner lid ^ m stays inside the lane's\n"); + s.push_str( + "// aligned run of 32 because m < 32. Control flow is uniform, so every work-item reaches every barrier.\n", + ); + s.push_str("#define IGNEUM_SHFL_XOR(dst, a, m) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid ^ (uint)(m))]; xk += 1u; }\n"); + s.push_str("#define IGNEUM_BCAST0(dst, a) { xch[(xk & 1u) * IGNEUM_GROUP + lid] = (a); barrier(CLK_LOCAL_MEM_FENCE); dst = xch[(xk & 1u) * IGNEUM_GROUP + (lid & ~31u)]; xk += 1u; }\n"); + s.push_str("#endif\n"); + s.push('\n'); + s.push_str(&hot_define(p)); + s.push_str("static inline uint splitmix32(uint x) {\n"); + s.push_str(" x ^= x >> 16; x *= 0x7feb352du;\n"); + s.push_str(" x ^= x >> 15; x *= 0x846ca68bu;\n"); + s.push_str(" x ^= x >> 16;\n"); + s.push_str(" return x;\n"); + s.push_str("}\n"); + s.push_str("// n is a literal in 1..31 at every call site. OpenCL rotate() rotates left by n modulo 32.\n"); + s.push_str("static inline uint rotl_imm(uint x, uint n) { return rotate(x, n); }\n"); + s.push_str("// Right rotation by n modulo 32 as a left rotation by (32 - n) modulo 32; n == 0 gives x.\n"); + s.push_str("static inline uint rotr_var(uint x, uint n) { return rotate(x, (0u - n) & 31u); }\n"); + s.push_str("static inline uint ds_elem(uint i, uint d0, uint d1) {\n"); + s.push_str(" uint x = i ^ d0;\n"); + s.push_str(DS_ELEM_BODY); + s.push('\n'); + if let Some(mp) = memhard { + s.push_str(&emit_memhard_core_layout(mp, CoreDialect::OpenCl, layout)); + s.push('\n'); + s.push_str("// Memory-hard dataset (MEMHARD.md). One work-item per cache segment; one work-item per 64-byte dataset item.\n"); + s.push_str("// The same constants as memhard.h in this pack (one emitter, three dialects).\n"); + s.push_str("__kernel void igneum_cache_fill(__global uint* cache, uint nSegments) {\n"); + s.push_str(" uint seg = (uint)get_global_id(0);\n"); + s.push_str(" if (seg < nSegments) mh_cache_segment(cache, seg);\n"); + s.push_str("}\n"); + if p.class.state { + s.push_str("// Class v5: the window's leaves (leaves.bin, IGNEUM_STATE_LEAVES x 16 words) and their count.\n"); + s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, __global const uint* leaves, uint nLeaves, uint nItems) {\n"); + } else { + s.push_str("__kernel void igneum_build(__global uint* ds, __global const uint* cache, uint nItems) {\n"); + } + s.push_str(" uint t = (uint)get_global_id(0);\n"); + s.push_str(" if (t < nItems) {\n"); + s.push_str(" uint s[16];\n"); + if p.class.state { + s.push_str(" mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s);\n"); + } else { + s.push_str(" mh_item(cache, t, s);\n"); + } + s.push_str(&build_store(layout, CoreDialect::OpenCl, "ds", "t")); + s.push_str(" }\n"); + s.push_str("}\n"); + if p.has_hot() { + s.push_str(&emit_hot_core(p, CoreDialect::OpenCl)); + s.push_str("// Hot table: one work-item per segment.\n"); + s.push_str("__kernel void igneum_hot_fill(__global uint* hot, uint nSegments) {\n"); + s.push_str(" uint seg = (uint)get_global_id(0);\n"); + s.push_str(" if (seg < nSegments) ht_segment(hot, seg);\n"); + s.push_str("}\n"); + } + s.push('\n'); + } else if p.has_hot() { + panic!("a hot-table pack needs the memory-hard dataset (the hot fill shares its ChaCha core)"); + } else { + s.push_str("// dataset[i] = ds_elem(i, d0, d1) for i < n. Same closed form as the Metal igneum_fill kernel.\n"); + s.push_str("__kernel void igneum_fill(__global uint* ds, uint n, uint d0, uint d1) {\n"); + s.push_str(" uint i = (uint)get_global_id(0);\n"); + s.push_str(" if (i < n) ds[i] = ds_elem(i, d0, d1);\n"); + s.push_str("}\n"); + s.push('\n'); + } + s.push_str("// One hash per work-item. IGNEUM_GROUP is a multiple of 32; lane = lid & 31 and every exchange stays inside the\n"); + s.push_str("// lane's own aligned run of 32 work-items, exactly like simd_shuffle_xor inside a 32-wide Metal SIMD group and\n"); + s.push_str("// __shfl_xor_sync inside a CUDA warp. Control flow is uniform (no branches at all).\n"); + s.push_str(&scratch_prelude(p, CoreDialect::OpenCl)); + let scratch_args = if p.has_scratch() { ", __global uint* scratch, uint groups, uint salt" } else { "" }; + let hot_args = if p.has_hot() { ", __global const uint* hot" } else { "" }; + s.push_str(&format!("IGNEUM_KERNEL_HASH void igneum_hash(__global const uint* ds, __global ulong* out, uint baseNonce, uint mask{hot_args}{scratch_args}) {{\n")); + let (setup, unit_loop) = persistent_prologue_parts(CoreDialect::OpenCl, p.class.scratch_words_per_lane()); + if p.has_scratch() { + s.push_str(&setup); + } else { + s.push_str(" uint gid = (uint)get_global_id(0);\n"); + } + s.push_str(" uint lid = (uint)get_local_id(0);\n"); + if p.has_scratch() { + s.push_str("#if IGNEUM_EXCHANGE == 0\n"); + s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); + s.push_str(" uint xk = 0u;\n"); + s.push_str("#else\n"); + s.push_str(" (void)lid;\n"); + s.push_str("#endif\n"); + s.push_str(&unit_loop); + s.push_str(" uint nonce = baseNonce + gid;\n"); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + } else { + s.push_str(" uint nonce = baseNonce + gid;\n"); + s.push_str(" uint r0, r1, r2, r3, r4, r5, r6, r7;\n"); + s.push_str("#if IGNEUM_EXCHANGE == 0\n"); + s.push_str(" IGNEUM_LOCAL_WORDS(xch, 2 * IGNEUM_GROUP);\n"); + s.push_str(" uint xk = 0u;\n"); + s.push_str("#else\n"); + s.push_str(" (void)lid;\n"); + s.push_str("#endif\n"); + } + if p.has_wide() { + s.push_str(" uint lane = lid & 31u;\n uint wmask = mask & ~31u;\n"); + } + for i in 0..8 { + s.push_str(&init_line(p, "uint", i)); + } + s.push_str(&format!("\n for (uint it = 0u; it < {ITERATIONS}u; ++it) {{\n uint sel = r0;\n")); + s.push_str(&opencl_instr_lines(p, dataset_log2)); + s.push_str(&shadow_block(p, CoreDialect::OpenCl)); + s.push_str(" }\n"); + s.push_str(" uint lo = r0 ^ rotl_imm(r1, 7u) ^ rotl_imm(r2, 14u) ^ rotl_imm(r3, 21u);\n"); + s.push_str(" uint hi = r4 ^ rotl_imm(r5, 9u) ^ rotl_imm(r6, 18u) ^ rotl_imm(r7, 27u);\n"); + s.push_str(" out[gid] = ((ulong)hi << 32) | (ulong)lo;\n"); + if p.has_scratch() { + s.push_str(" }\n"); + } + s.push_str("}\n"); + s.push('\n'); + s.push_str("#if IGNEUM_EXCHANGE != 0\n"); + s.push_str("// Reports the sub-group size this device uses for a work-group of IGNEUM_GROUP items. host.c runs it only when the\n"); + s.push_str("// per-kernel query (clGetKernelSubGroupInfoKHR on igneum_hash) is unavailable; that query is preferred because a\n"); + s.push_str("// compiler may pick a different wave width per kernel (RDNA: wave32 or wave64). See WAVEFRONT.md.\n"); + s.push_str("IGNEUM_KERNEL_HASH void igneum_probe_subgroup(__global uint* out) {\n"); + s.push_str(" if (get_local_id(0) == 0u) { out[0] = get_sub_group_size(); out[1] = get_num_sub_groups(); }\n"); + s.push_str("}\n"); + s.push_str("#endif\n"); + s +} + +fn mask_for(dataset_log2: u32) -> u32 { + if dataset_log2 >= 32 { + u32::MAX + } else { + (1u32 << dataset_log2) - 1 + } +} + +const STDINT_BLOCK: &str = "#ifdef __cplusplus\n#include \n#else\n#include \n#endif\n"; + +/// program.h (`generateProgramHeader`). +pub fn program_header(p: &Program, day: &str, ds: &DatasetSource) -> String { + let key = &ds.key; + let dataset_log2 = ds.log2_words; + let memhard = ds.memhard().map(|m| &m.params); + let mask = mask_for(dataset_log2); + let mut s = String::with_capacity(2600); + s.push_str(&generated_by(&p.seed_string)); + s.push_str("// Program metadata for host.cu plus the launch wrappers defined in kernel.cu.\n"); + s.push_str("// Also included by proto-opencl/host.c (C99), which defines IGNEUM_NO_CUDA first and reads only the macros.\n"); + s.push_str("#pragma once\n"); + s.push_str(STDINT_BLOCK); + s.push_str("#ifndef IGNEUM_NO_CUDA\n#include \n#endif\n"); + s.push('\n'); + s.push_str(&format!("#define IGNEUM_SEED_STRING {}\n", jstr(&p.seed_string))); + s.push_str(&format!("#define IGNEUM_SEED_BYTES_HEX {}\n", jstr(&hex_bytes(&p.seed_bytes)))); + s.push_str(&format!("#define IGNEUM_GENERATOR {}\n", p.generator)); + s.push_str(&format!("#define IGNEUM_PROGRAM_ATTEMPT {}\n", p.attempt)); + s.push_str(&format!("#define IGNEUM_PROGRAM_ID {}\n", hex64(p.program_id()))); + s.push_str(&format!("#define IGNEUM_DAY_STRING {}\n", jstr(day))); + s.push_str(&format!("#define IGNEUM_DAY_BYTES_HEX {}\n", jstr(&hex_bytes(&ds.key_bytes)))); + s.push_str(&format!("#define IGNEUM_DAY0 {}\n", hex(key[0]))); + s.push_str(&format!("#define IGNEUM_DAY1 {}\n", hex(key[1]))); + s.push_str(&format!("#define IGNEUM_DATASET_LOG2 {dataset_log2}\n")); + s.push_str(&format!("#define IGNEUM_MASK {}\n", hex(mask))); + s.push_str("#define IGNEUM_LANES 32\n"); + s.push_str(&format!("#define IGNEUM_ITERATIONS {ITERATIONS}\n")); + s.push_str(&format!("#define IGNEUM_INSTR_COUNT {INSTR_COUNT}\n")); + s.push_str(&format!("#define IGNEUM_LOADS_PER_HASH {}\n", p.loads_per_hash())); + s.push_str(&format!("#define IGNEUM_WIDE_LOADS_PER_HASH {}\n", p.wide_loads_per_hash())); + s.push_str(&format!("#define IGNEUM_OP_MIX {}\n", jstr(&p.op_mix()))); + s.push_str(&program_class_header_lines(p)); + s.push_str(&class_header_lines(p)); + s.push_str(&state_header_lines(ds)); + s.push_str(&scratch_header_lines(p)); + s.push_str(&era_header_lines(p)); + s.push_str(&hot_header_lines(p)); + s.push_str(&shadow_header_lines(p)); + s.push_str("// 0 = closed-form dataset (ds_elem), 1 = memory-hard cache construction (MEMHARD.md, memhard.h)\n"); + s.push_str(&format!("#define IGNEUM_DATASET_MODE {}\n", if memhard.is_some() { 1 } else { 0 })); + s.push('\n'); + s.push_str(&format!("#define IGNEUM_SEEDW_INIT {{ {} }}\n", join_hex(&p.seed))); + if let Some(mp) = memhard { + s.push_str(&format!("#define IGNEUM_KEY_INIT {{ {} }}\n", join_hex(&mp.key))); + s.push_str(&format!("#define IGNEUM_CACHE_LOG2_WORDS {}\n", mp.shape.cache_log2_words)); + s.push_str(&format!("#define IGNEUM_CACHE_SEGMENT_LOG2_LINES {CACHE_SEGMENT_LOG2_LINES}\n")); + s.push_str(&format!("#define IGNEUM_CACHE_SEGMENTS {}u\n", mp.shape.cache_segments())); + s.push_str(&format!("#define IGNEUM_ITEM_ROUNDS {ITEM_ROUNDS}\n")); + if mp.shape.mixer_mult != 1 { + s.push_str(&format!("#define IGNEUM_MIXER_MULT {} // mixer applications per round and after the last read (class v3, docs/plans/mixer-x4.md)\n", mp.shape.mixer_mult)); + } + if let Some(dp) = &mp.derive { + s.push_str(&format!("#define IGNEUM_DERIVE_LEN {} // instructions per round program of the day's item-derivation program (Counter ASIC 3.0 item 2; memhard.h mh_round_0..8)\n", dp.len)); + s.push_str(&format!("#define IGNEUM_DERIVE_ATTEMPT {}\n", dp.attempt)); + s.push_str(&format!("#define IGNEUM_DERIVE_FINGERPRINT {}\n", hex64(dp.fingerprint()))); + s.push_str(&format!("#define IGNEUM_DERIVE_INSTRS_PER_ITEM {}\n", dp.instr_count())); + s.push_str(&format!("#define IGNEUM_DERIVE_GPU_OPS_PER_ITEM {}\n", dp.gpu_ops())); + s.push_str(&format!("#define IGNEUM_DERIVE_CHIP_OPS_PER_ITEM {}\n", dp.chip_ops())); + s.push_str(&format!("#define IGNEUM_DERIVE_MULS_PER_ITEM {}\n", dp.muls())); + } + s.push_str(&format!( + "#define IGNEUM_MIX_ROT_INIT {{ {} }}\n", + mp.rot.iter().map(|r| format!("{r}u")).collect::>().join(", ") + )); + s.push_str(&format!("#define IGNEUM_MIX_MUL_INIT {{ {} }}\n", join_hex(&mp.mul))); + s.push_str(&format!("#define IGNEUM_MIX_RC_INIT {{ {} }}\n", join_hex(&mp.rc))); + s.push('\n'); + s.push_str("#ifndef IGNEUM_NO_CUDA\n"); + s.push_str("// Defined in kernel.cu. All launch on the default stream and return cudaGetLastError().\n"); + s.push_str("cudaError_t igneum_launch_cache_fill(uint32_t* cache, uint32_t nSegments);\n"); + if p.class.state { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems);\n"); + } else { + s.push_str("cudaError_t igneum_launch_build(uint32_t* ds, const uint32_t* cache, uint32_t nItems);\n"); + } + if p.has_hot() { + s.push_str("cudaError_t igneum_launch_hot_fill(uint32_t* hot, uint32_t nSegments);\n"); + } + } else { + s.push_str("#ifndef IGNEUM_NO_CUDA\n"); + s.push_str("// Defined in kernel.cu. Both launch on the default stream and return cudaGetLastError().\n"); + s.push_str("cudaError_t igneum_launch_fill(uint32_t* ds, uint32_t nWords, uint32_t d0, uint32_t d1);\n"); + } + let hot_decl = if p.has_hot() { " const uint32_t* hot," } else { "" }; + if p.has_scratch() { + s.push_str(&format!("cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n")); + s.push_str(" uint32_t nonces, uint32_t blockWarps, uint32_t* scratch, uint32_t warps, uint32_t salt);\n"); + } else { + s.push_str(&format!( + "cudaError_t igneum_launch_hash(const uint32_t* ds, uint64_t* out, uint32_t baseNonce, uint32_t mask,{hot_decl}\n", + )); + s.push_str(" uint32_t nonces, uint32_t blockWarps);\n"); + } + s.push_str("cudaError_t igneum_hash_info(int* numRegs, int* blocksPerSM, uint32_t blockWarps);\n"); + s.push_str("#endif\n"); + s +} + +/// memhard.h (`generateMemhardHeader`): the core in CUDA C++, compiled for host and device. +pub fn cuda_memhard_header(p: &Program, mp: &MixParams) -> String { + let mut s = String::with_capacity(6000); + s.push_str(&generated_by(&p.seed_string)); + s.push_str("// Memory-hard dataset core, the same text that the Mac's Metal kernels and CPU verifier were checked against.\n"); + s.push_str( + "// Included by kernel.cu (device), host.cu (host reference) and proto-opencl/host.c (C99 host reference).\n", + ); + s.push_str("// See proto-metal/MEMHARD.md for the construction. kernel.cl carries the same text in OpenCL C.\n"); + s.push_str("#pragma once\n"); + s.push_str(STDINT_BLOCK); + s.push_str("#if defined(__CUDACC__)\n"); + s.push_str("#define IGNEUM_HD __host__ __device__ __forceinline__\n"); + s.push_str("#elif defined(_MSC_VER) && !defined(__cplusplus)\n"); + s.push_str("#define IGNEUM_HD static __inline\n"); + s.push_str("#else\n"); + s.push_str("#define IGNEUM_HD static inline\n"); + s.push_str("#endif\n"); + s.push_str(&emit_memhard_core_layout(mp, CoreDialect::Cuda, p.class.layout())); + if p.has_hot() { + s.push('\n'); + s.push_str(&emit_hot_core(p, CoreDialect::Cuda)); + } + s +} + +/// The self-test values a pack carries beside the 96 hashes. +#[derive(Clone, Debug, Default, PartialEq, Eq)] +pub struct PackVectors { + /// dataset[0..15] + pub head: Vec, + /// dataset[MASK] + pub last: u32, + /// 64 sampled dataset indices and their values + pub sample_idx: Vec, + pub sample_val: Vec, + /// cache[0..15] (memory-hard only) + pub cache_head: Vec, + /// the last cache line (memory-hard only) + pub cache_last: Vec, + /// FNV-1a 64 over the whole cache (memory-hard only) + pub cache_fnv: u64, + /// The cache is 2^cache_log2_words words (memory-hard only; 26 under version 2) + pub cache_log2_words: u32, + /// Hot table (hot packs only): head line, last line, FNV-1a 64 over the whole table + pub has_hot: bool, + pub hot_head: Vec, + pub hot_last: Vec, + pub hot_fnv: u64, +} + +/// The base nonces of the three vector warps every pack carries. +pub const PACK_VECTOR_BASES: [u32; 3] = [0, 4096, 1_000_000]; + +/// The 64 sampled dataset indices: SplitMix64 seeded with "mhsample", low 32 bits masked. +pub fn sample_indices(mask: u32) -> Vec { + let mut sr = SplitMix64::new(0x6d68_7361_6d70_6c65); + (0..64).map(|_| (sr.next() as u32) & mask).collect() +} + +/// vectors.h (`generateVectorsHeader`). +pub fn vectors_header( + p: &Program, + bases: &[u32], + outs: &[[u64; 32]], + v: &PackVectors, + mask: u32, + source: &str, + memhard: bool, +) -> String { + let mut s = String::with_capacity(6000); + s.push_str(&generated_by(&p.seed_string)); + s.push_str(&format!("// Expected outputs: {source}\n")); + s.push_str("#pragma once\n"); + s.push_str(STDINT_BLOCK); + s.push('\n'); + s.push_str(&format!("#define IGNEUM_VEC_WARPS {}\n", bases.len())); + s.push_str(&format!( + "static const uint32_t IGNEUM_VEC_BASE[IGNEUM_VEC_WARPS] = {{ {} }};\n", + bases.iter().map(|b| format!("{b}u")).collect::>().join(", ") + )); + s.push_str("static const uint64_t IGNEUM_VEC_OUT[IGNEUM_VEC_WARPS][32] = {\n"); + for (i, o) in outs.iter().enumerate() { + s.push_str(&format!(" {{ // base nonce {}\n", bases[i])); + for row in 0..4 { + s.push_str(" "); + s.push_str(&(0..8).map(|c| hex64(o[row * 8 + c])).collect::>().join(", ")); + s.push_str(if row == 3 { "\n" } else { ",\n" }); + } + s.push_str(if i == outs.len() - 1 { " }\n" } else { " },\n" }); + } + s.push_str("};\n"); + s.push('\n'); + s.push_str(&format!("// Dataset self-test: dataset[0..15] and dataset[IGNEUM_MASK] ({mask}).\n")); + s.push_str("static const uint32_t IGNEUM_DS_HEAD[16] = {\n"); + s.push_str(&format!(" {},\n", join_hex(&v.head[..8]))); + s.push_str(&format!(" {}\n", join_hex(&v.head[8..16]))); + s.push_str("};\n"); + s.push_str(&format!("static const uint32_t IGNEUM_DS_LAST_INDEX = {mask}u;\n")); + s.push_str(&format!("static const uint32_t IGNEUM_DS_LAST = {};\n", hex(v.last))); + s.push_str("// 64 sampled dataset words (index, value) computed on the Mac.\n"); + s.push_str(&format!("#define IGNEUM_DS_SAMPLES {}\n", v.sample_idx.len())); + s.push_str("static const uint32_t IGNEUM_DS_SAMPLE_INDEX[IGNEUM_DS_SAMPLES] = {\n"); + s.push_str(&format!(" {}\n", v.sample_idx.iter().map(|i| format!("{i}u")).collect::>().join(", "))); + s.push_str("};\n"); + s.push_str("static const uint32_t IGNEUM_DS_SAMPLE_VALUE[IGNEUM_DS_SAMPLES] = {\n"); + s.push_str(&format!(" {}\n", join_hex(&v.sample_val))); + s.push_str("};\n"); + if memhard { + s.push_str(&format!( + "// Cache self-test (memory-hard mode): cache[0..15], the last 16 words, and FNV-1a 64 over all 2^{} words.\n", + v.cache_log2_words + )); + s.push_str("static const uint32_t IGNEUM_CACHE_HEAD[16] = {\n"); + s.push_str(&format!(" {},\n", join_hex(&v.cache_head[..8]))); + s.push_str(&format!(" {}\n", join_hex(&v.cache_head[8..16]))); + s.push_str("};\n"); + s.push_str("static const uint32_t IGNEUM_CACHE_LAST[16] = {\n"); + s.push_str(&format!(" {},\n", join_hex(&v.cache_last[..8]))); + s.push_str(&format!(" {}\n", join_hex(&v.cache_last[8..16]))); + s.push_str("};\n"); + s.push_str(&format!("static const uint64_t IGNEUM_CACHE_FNV64 = {};\n", hex64(v.cache_fnv))); + } + if v.has_hot { + s.push_str("// Hot table self-test (docs/plans/hot-table.md): hot[0..15], the last 16 words, and FNV-1a 64 over all IGNEUM_HOT_WORDS words.\n"); + s.push_str("static const uint32_t IGNEUM_HOT_HEAD[16] = {\n"); + s.push_str(&format!(" {},\n", join_hex(&v.hot_head[..8]))); + s.push_str(&format!(" {}\n", join_hex(&v.hot_head[8..16]))); + s.push_str("};\n"); + s.push_str("static const uint32_t IGNEUM_HOT_LAST[16] = {\n"); + s.push_str(&format!(" {},\n", join_hex(&v.hot_last[..8]))); + s.push_str(&format!(" {}\n", join_hex(&v.hot_last[8..16]))); + s.push_str("};\n"); + s.push_str(&format!("static const uint64_t IGNEUM_HOT_FNV64 = {};\n", hex64(v.hot_fnv))); + } + s +} + +/// program.json (`generateProgramJSON`). Valid JSON (see the module note about the `"item"` line). +pub fn program_json(p: &Program, day: &str, ds: &DatasetSource) -> String { + let key = &ds.key; + let dataset_log2 = ds.log2_words; + let memhard = ds.memhard().map(|m| &m.params); + let mask = mask_for(dataset_log2); + let mut s = String::with_capacity(14000); + s.push_str("{\n"); + s.push_str(" \"format\": \"igneum-program-pack-3\",\n"); + s.push_str(&format!(" \"generator\": {},\n", p.generator)); + s.push_str(&format!(" \"attempt\": {},\n", p.attempt)); + s.push_str(&format!(" \"program_id\": {},\n", jhex64(p.program_id()))); + s.push_str(" \"program_id_derivation\": \"FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32\",\n"); + s.push_str(&format!( + " \"dataset_mode\": {},\n", + jstr(if memhard.is_some() { "memory-hard" } else { "closed-form" }) + )); + s.push_str(&format!(" \"seed\": {},\n", jstr(&p.seed_string))); + s.push_str(&format!(" \"seed_bytes\": {},\n", jstr(&hex_bytes(&p.seed_bytes)))); + s.push_str(&format!(" \"seed_words\": [{}],\n", join_jhex(&p.seed))); + s.push_str(" \"seed_derivation\": \"seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32\",\n"); + s.push_str(&format!(" \"generator_rule\": \"version {GENERATOR_VERSION}: exactly {LOAD_SLOTS} load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried\",\n")); + s.push_str(" \"lanes\": 32,\n"); + s.push_str(" \"registers\": 8,\n"); + s.push_str(&format!(" \"iterations\": {ITERATIONS},\n")); + s.push_str(&format!(" \"instruction_count\": {INSTR_COUNT},\n")); + s.push_str(&format!(" \"loads_per_hash\": {},\n", p.loads_per_hash())); + if p.program_class() != ProgramClass::V2 { + s.push_str(&format!(" \"program_class\": {},\n", jstr(p.program_class().name()))); + if p.program_class() == ProgramClass::V4 { + s.push_str(&format!(" \"sub_version\": {},\n", PROGRAM_SUBVERSION_V4)); + } + if let Some(era) = &p.era_bytes { + s.push_str(&format!(" \"era_seed_bytes\": {},\n", jstr(&hex_bytes(era)))); + } + } + if let Some(l) = ds.leaves() { + s.push_str(" \"state\": {\n"); + s.push_str(&format!(" \"block\": {},\n", jstr(&hex_bytes(&l.block)))); + s.push_str(&format!(" \"block_number\": {},\n", l.number)); + s.push_str(&format!(" \"root\": {},\n", jstr(&hex_bytes(&l.root)))); + s.push_str(&format!(" \"leaves\": {},\n", l.n())); + s.push_str(&format!(" \"records\": {},\n", l.records_total)); + s.push_str(&format!(" \"sampled\": {},\n", l.sampled)); + s.push_str(&format!(" \"leaves_fnv1a64\": {},\n", jhex64(l.fnv1a64()))); + s.push_str(" \"leaf_derivation\": \"leaves[i] = Blake2b-512('igneum-sd1/' || root || i_le32 || record_i) as 16 little-endian words; item t XORs leaves[t mod leaves] into its 16 initial words before the first mixer\",\n"); + s.push_str(" \"file\": \"leaves.bin\"\n"); + s.push_str(" },\n"); + } + if !p.class.is_v2() { + let c = p.width_counts(); + s.push_str(&format!(" \"load_class\": {},\n", jstr(&p.class.name()))); + if p.class.derive_len != 0 { + s.push_str(&format!(" \"derive_len\": {},\n", p.class.derive_len)); + s.push_str(" \"derive\": \"Counter ASIC 3.0 item 2 (6 October 2026, docs/plans/counter-asic-3-derivation.md; a prototype, not class v3): the nine mixer slots of the item derivation each run a straight-line program of derive_len instructions drawn from the day key stream after the 40 mixer draws, four draws per instruction (op roll below(100), destination roll below(15), third-register roll below(14), the immediate next()); every instruction reads the register the previous one wrote (s[0] first) and writes another; twelve forms, each a bijection on the state; the 8 dependent cache reads per item unchanged; the acceptance test of derive.rs (every register written per round program, 8 distinct rotations, the x8 mixer's operation and multiply counts as floors) rejects a draw and the next attempt continues the stream\",\n"); + } + if p.class.mixer_mult != 1 || p.class.growth { + s.push_str(&format!(" \"mixer_mult\": {},\n", p.class.mixer_mult)); + s.push_str(&format!(" \"cache_growth\": {},\n", p.class.growth)); + s.push_str(&format!(" \"mixer\": \"class v3 (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): every mixer application of the item derivation is {} applications with round keys (r * {} + j + 1) * 0x9E3779B9, the 8 dependent cache reads per item unchanged; cache growth rule option C: cache words = 2^(26 + doublings(day)), dataset words = 2^(genesis_log2 + doublings(day)), doublings(day) = floor(log2(1 + day / 1460)) for day = days since genesis\",\n", p.class.mixer_mult, p.class.mixer_mult)); + } + s.push_str(&format!(" \"load_slots\": {},\n", p.class.load_slots)); + s.push_str(&format!(" \"load_mix_percent_4_16_64\": [{}, {}, {}],\n", p.class.mix[0], p.class.mix[1], p.class.mix[2])); + s.push_str(&format!(" \"load_width_counts_4_16_64\": [{}, {}, {}],\n", c[0], c[1], c[2])); + s.push_str(&format!(" \"bytes_per_hash\": {},\n", p.bytes_per_hash())); + if p.has_scratch() { + s.push_str(&format!(" \"scratch_ops_per_hash\": {},\n", p.scratch_ops_per_hash())); + s.push_str(&format!(" \"scratch_kib_per_warp\": {},\n", p.class.scratch_kb)); + s.push_str(&format!(" \"scratch\": \"variant 5 (measurement only): persistent warps; a {kb} KiB scratch per warp of {slots} 16-byte slots per lane (lane-major); slot = src & 0x{smask:x}; a slot reads as its fill (scratch_fill(seed words, unit base nonce, lane, slot, j) = splitmix32(((base + lane) ^ seed[j]) + slot * 0x9e3779b1 + (j + 1) * 0x85ebca77), j in 0..2) until the unit writes it; read w0 w1 w2 (behind a per-unit tag on the GPU), x = fold(dst, w0, w1, w2), dst = x, rewrite (x ^ w1, rotl(x, 7) ^ w2, x + w0)\",\n", kb = p.class.scratch_kb, slots = p.class.scratch_slots_per_lane(), smask = p.class.scratch_slot_mask())); + } + s.push_str(&format!(" \"wide_load\": \"read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, {FOLD_ROT}) * 0x{FOLD_MUL:08x}) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots\",\n")); + if let Some(e) = p.class.era { + s.push_str(" \"era\": {\n"); + s.push_str(&format!(" \"label\": {},\n", jstr(&e.label()))); + s.push_str(&format!(" \"seed_words\": [{}],\n", join_jhex(&e.words))); + s.push_str(" \"draw\": \"docs/plans/era-layout.md 1.1: SplitMix64 seeded with seed_words[0] | seed_words[1] << 32 of seed_words_from_bytes('igneum-era/' || n_le64 || E_n); width = allowed[below(|allowed|)], stride_mul = low32(next()) | 1, stride_rot = 1 + below(31), then four next() draws for a partial Fisher-Yates over positions log2(W)..15 of which 4 - log2(W) are used\",\n"); + s.push_str(&format!(" \"allowed_widths\": [{}],\n", e.allowed_set().iter().map(|w| w.to_string()).collect::>().join(", "))); + s.push_str(&format!(" \"width_words\": {},\n", e.width_words)); + s.push_str(&format!(" \"stride_mul\": {},\n", jhex(e.stride_mul))); + s.push_str(&format!(" \"stride_rot\": {},\n", e.stride_rot)); + s.push_str(&format!(" \"interleave\": [{}, {}, {}, {}],\n", e.pos[0], e.pos[1], e.pos[2], e.pos[3])); + s.push_str(" \"address\": \"y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words\",\n"); + s.push_str(" \"windows\": \"per instruction, after the width roll: win = below(3), off = low32(next()) & (2^win - 1); used on a load slot (the instruction's win and off fields)\",\n"); + s.push_str(" \"dataset_word\": \"dataset[w] = item(t(w))[j(w)]: j(w) gathers the bits of w at the interleave positions, t(w) is w with those bits removed\",\n"); + s.push_str(" \"program_id_suffix\": \"'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]\"\n"); + s.push_str(" },\n"); + } + if let Some(h) = p.class.hot { + let hk = hot_key(&p.seed_bytes); + s.push_str(&format!( + " \"hot_table\": {{\"mb\": {}, \"words\": {}, \"segments\": {}, \"slots\": {}, \"form\": {}, \"dataset_slots\": {}, \"hot_loads_per_hash\": {}, \"key\": [{}], \"key_derivation\": \"seed_words_from_bytes('igneum-hot/' || seed_bytes), the same for every attempt of the epoch\", \"tag\": [{}], \"chain\": \"the cache chain of dataset.cache with the hot key and tag: in_j = prev ^ (sigma || key || seg || j || tag); line_j = block(in_j); prev_0 = 0\", \"load\": \"dst = dst ^ hot[mulhi(src, words)] (the high 32 bits of the 64-bit product; the index lies in [0, words) for any size)\", \"slots_rule\": \"the first k load slots in the Fisher-Yates draw order after the scratch slots (a uniform k-subset); no extra draw, so with version 2 widths and no scratch the program is the version 2 program with k loads redirected\", \"acceptance_stand_in\": \"dataset_elem(mulhi(src, words), seed_words[2], seed_words[3])\", \"program_id\": \"the read-width id with 'hot/' || mb || k appended\", \"spec\": \"docs/plans/hot-table.md\"}},\n", + h.mb, + hot_words(h.mb as u32), + hot_segments(h.mb as u32), + h.k, + jstr(if h.added { "added: k load slots added beside the class's, the dataset loads unchanged" } else { "replaced: k of the class's load slots read the table" }), + p.class.dataset_slots(), + p.hot_loads_per_hash(), + join_jhex(&hk), + join_jhex(&HOT_TAG) + )); + } + } + s.push_str(&format!( + " \"op_mix\": {{{}}},\n", + p.histogram().iter().map(|(n, c)| format!("{}: {c}", jstr(n))).collect::>().join(", ") + )); + s.push_str(" \"register_init\": \"for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]\",\n"); + s.push_str(" \"splitmix32\": \"x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16\",\n"); + s.push_str( + " \"iteration\": \"sel = r0 sampled once at the top of each iteration, then all instructions in order\",\n", + ); + s.push_str(" \"output\": \"lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo\",\n"); + s.push_str(" \"op_semantics\": {\n"); + s.push_str(" \"add\": \"dst = dst + src + (bit `bit` of sel ? imm2 : imm)\",\n"); + s.push_str(" \"sub\": \"dst = dst - src\",\n"); + s.push_str(" \"mul\": \"dst = dst * src (low 32)\",\n"); + s.push_str(" \"mulhi\": \"dst = high 32 bits of dst * src\",\n"); + s.push_str(" \"xor\": \"dst = dst ^ src\",\n"); + s.push_str(" \"or\": \"dst = dst | src\",\n"); + s.push_str(" \"rotl\": \"dst = rotl(dst, rot), rot in 1..31\",\n"); + s.push_str(" \"rotr\": \"dst = rotr(dst, src & 31)\",\n"); + s.push_str(" \"mad\": \"dst = src * src2 + dst\",\n"); + s.push_str( + " \"shfl\": \"dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp\",\n", + ); + s.push_str(" \"load\": \"dst = dst ^ dataset[src & dataset.mask]\",\n"); + s.push_str(" \"wload\": \"base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)\""); + if p.has_hot() { + s.push_str(",\n \"hot\": \"dst = dst ^ hot[mulhi(src, hot_table.words)] (hot-table experiment)\""); + } + s.push('\n'); + s.push_str(" },\n"); + s.push_str(" \"dataset\": {\n"); + s.push_str(&format!(" \"log2_words\": {dataset_log2},\n")); + s.push_str(&format!(" \"bytes\": {},\n", 1u64 << (dataset_log2 as u64 + 2))); + s.push_str(&format!(" \"mask\": {},\n", jhex(mask))); + s.push_str(&format!(" \"day\": {},\n", jstr(day))); + s.push_str(&format!(" \"day_bytes\": {},\n", jstr(&hex_bytes(&ds.key_bytes)))); + s.push_str(" \"day_words_from\": \"seed_words_from_bytes(day_bytes)\",\n"); + s.push_str(&format!(" \"d0\": {},\n", jhex(key[0]))); + s.push_str(&format!(" \"d1\": {},\n", jhex(key[1]))); + if let Some(mp) = memhard { + s.push_str(" \"mode\": \"memory-hard\",\n"); + s.push_str(" \"spec\": \"proto-metal/MEMHARD.md\",\n"); + s.push_str(&format!(" \"key\": [{}],\n", join_jhex(&mp.key))); + s.push_str(" \"key_derivation\": \"the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]\",\n"); + let shape = &mp.shape; + s.push_str(&format!( + " \"cache\": {{\"log2_words\": {}, \"bytes\": {}, \"line_words\": 16, \"segment_lines\": {CACHE_LINES_PER_SEGMENT}, \"segments\": {}, \"block\": \"ChaCha{CHACHA_ROUNDS} core + feed-forward, rotations 16 12 8 7\", \"sigma\": [{}], \"tag\": [{}], \"chain\": \"in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0\"}},\n", + shape.cache_log2_words, + shape.cache_words() as u64 * 4, + shape.cache_segments(), + join_jhex(&CHACHA_SIGMA), + join_jhex(&CACHE_TAG) + )); + s.push_str(&format!( + " \"mixer\": {{\"draw\": \"SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())\", \"rot\": [{}], \"mul\": [{}], \"rc\": [{}], \"round\": \"for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]\", \"quarter_round\": \"a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)\"}},\n", + mp.rot.iter().map(|r| r.to_string()).collect::>().join(", "), + join_jhex(&mp.mul), + join_jhex(&mp.rc) + )); + // The Swift writes jhex(cacheLineMask) here, which breaks the JSON. We write the bare literal. + if let Some(dp) = &mp.derive { + s.push_str(&format!( + " \"derive_len\": {},\n \"derive_attempt\": {},\n \"derive_fingerprint\": {},\n \"derive_op_mix\": {},\n \"derive_instrs_per_item\": {},\n \"derive_gpu_ops_per_item\": {},\n \"derive_chip_ops_per_item\": {},\n \"derive_muls_per_item\": {},\n", + dp.len, dp.attempt, jhex64(dp.fingerprint()), jstr(&dp.op_mix()), dp.instr_count(), dp.gpu_ops(), dp.chip_ops(), dp.muls() + )); + s.push_str(&format!( + " \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..{}: s = P_r(s); line = s[0] & 0x{:08x}; s[i] ^= cache[line * 16 + i]; then s = P_{ITEM_ROUNDS}(s); item(t) = s; P_r is round program r below (c = the register the previous instruction wrote, s[0] first): add d += c; sub d -= c; xor d ^= c; mul d *= (c | 1); rot d = rotl(d, k) + c; xrot d = rotl(d ^ c, k); addc d += c + imm; xorc d ^= c ^ imm; mulc d = (d ^ c) * imm; mulc2 d = d * imm + c; andx d ^= (c & b); orx d += (c | b)\",\n", + ITEM_ROUNDS - 1, + shape.cache_line_mask() + )); + s.push_str(" \"programs\": [\n"); + for (r, prog) in dp.rounds.iter().enumerate() { + s.push_str(" ["); + s.push_str(&prog.iter().map(|i| jstr(&derive_instr_line(i))).collect::>().join(", ")); + s.push_str(if r + 1 < dp.rounds.len() { "],\n" } else { "]\n" }); + } + s.push_str(" ],\n"); + } else if shape.mixer_mult == 1 { + s.push_str(&format!( + " \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..{}: s = M_r(s); line = s[0] & 0x{:08x}; s[i] ^= cache[line * 16 + i]; then s = M_{ITEM_ROUNDS}(s); item(t) = s\",\n", + ITEM_ROUNDS - 1, + shape.cache_line_mask() + )); + } else { + let m = shape.mixer_mult; + s.push_str(&format!( + " \"mixer_mult\": {m},\n \"item\": \"s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..{}: for j in 0..{}: s = M(s, rk = (r * {m} + j + 1) * 0x9E3779B9); line = s[0] & 0x{:08x}; s[i] ^= cache[line * 16 + i]; then for j in 0..{}: s = M(s, rk = ({} + j + 1) * 0x9E3779B9); item(t) = s\",\n", + ITEM_ROUNDS - 1, + m - 1, + shape.cache_line_mask(), + m - 1, + ITEM_ROUNDS as u32 * m + )); + } + s.push_str(" \"word\": \"dataset[w] = item(w >> 4)[w & 15]\"\n"); + } else { + s.push_str(" \"mode\": \"closed-form\",\n"); + s.push_str(" \"formula\": \"x = i ^ d0; x *= 0x9E3779B1; x ^= x>>15; x += d1; x *= 0x85EBCA77; x ^= x>>13; x *= 0xC2B2AE3D; x ^= x>>16 (all mod 2^32)\"\n"); + } + s.push_str(" },\n"); + if let Some(sh) = p.class.shadow { + s.push_str(&format!( + " \"shadow\": {{\"instrs\": {}, \"reps\": {}, \"instrs_per_hash\": {}, \"op_mix\": {{{}}}, \"rule\": \"Counter ASIC 3.0 item 8 (docs/analysis/latency-shadow-2026-10-06.md): after the 64 base instructions the program stream draws instrs more ALU instructions (the non-load table, nine draws each, the source as on an ALU slot); the block runs reps times at the end of every iteration with the iteration's sel; the acceptance rule interprets the base program only\", \"program_id_suffix\": \"'shadow/' || instrs_le16 || reps_le16\", \"instructions\": [\n", + sh.instrs, + sh.reps, + p.shadow_instrs_per_hash(), + p.shadow_histogram().iter().map(|(n, c)| format!("{}: {c}", jstr(n))).collect::>().join(", ") + )); + let n = p.shadow.len(); + for (k, ins) in p.shadow.iter().enumerate() { + s.push_str(&format!( + " {{\"i\": {k}, \"op\": {}, \"dst\": {}, \"src\": {}, \"src2\": {}, \"imm\": {}, \"imm2\": {}, \"rot\": {}, \"bit\": {}, \"mask\": {}}}{}\n", + jstr(ins.op.name()), + ins.dst, + ins.src, + ins.src2, + jhex(ins.imm), + jhex(ins.imm2), + ins.rot, + ins.bit, + ins.mask, + if k + 1 < n { "," } else { "" } + )); + } + s.push_str(" ]},\n"); + } + s.push_str(" \"instructions\": [\n"); + let n = p.instrs.len(); + for (k, ins) in p.instrs.iter().enumerate() { + if !p.class.is_v2() { + let era_fields = if p.class.era.is_some() { format!(", \"win\": {}, \"off\": {}", ins.win, ins.off) } else { String::new() }; + s.push_str(&format!( + " {{\"i\": {k}, \"op\": {}, \"dst\": {}, \"src\": {}, \"src2\": {}, \"imm\": {}, \"imm2\": {}, \"rot\": {}, \"bit\": {}, \"mask\": {}, \"width\": {}{era_fields}}}", + jstr(ins.op.name()), + ins.dst, + ins.src, + ins.src2, + jhex(ins.imm), + jhex(ins.imm2), + ins.rot, + ins.bit, + ins.mask, + ins.width + )); + s.push_str(if k + 1 < n { ",\n" } else { "\n" }); + continue; + } + s.push_str(&format!( + " {{\"i\": {k}, \"op\": {}, \"dst\": {}, \"src\": {}, \"src2\": {}, \"imm\": {}, \"imm2\": {}, \"rot\": {}, \"bit\": {}, \"mask\": {}}}", + jstr(ins.op.name()), + ins.dst, + ins.src, + ins.src2, + jhex(ins.imm), + jhex(ins.imm2), + ins.rot, + ins.bit, + ins.mask + )); + s.push_str(if k == n - 1 { "\n" } else { ",\n" }); + } + s.push_str(" ]\n}\n"); + s +} + +/// vectors.json (`generateVectorsJSON`). +pub fn vectors_json( + p: &Program, + day: &str, + dataset_log2: u32, + bases: &[u32], + outs: &[[u64; 32]], + v: &PackVectors, + mask: u32, + source: &str, + memhard: bool, +) -> String { + let mut s = String::with_capacity(6500); + s.push_str("{\n"); + s.push_str(&format!(" \"seed\": {},\n", jstr(&p.seed_string))); + s.push_str(&format!(" \"day\": {},\n", jstr(day))); + s.push_str(&format!(" \"dataset_mode\": {},\n", jstr(if memhard { "memory-hard" } else { "closed-form" }))); + s.push_str(&format!(" \"dataset_log2_words\": {dataset_log2},\n")); + s.push_str(&format!(" \"mask\": {},\n", jhex(mask))); + s.push_str(" \"lanes\": 32,\n"); + s.push_str(&format!(" \"source\": {},\n", jstr(source))); + s.push_str(" \"warps\": [\n"); + for (i, o) in outs.iter().enumerate() { + s.push_str(&format!(" {{\"base_nonce\": {}, \"expected\": [\n", bases[i])); + for row in 0..4 { + s.push_str(" "); + s.push_str(&(0..8).map(|c| jhex64(o[row * 8 + c])).collect::>().join(", ")); + s.push_str(if row == 3 { "\n" } else { ",\n" }); + } + s.push_str(if i == outs.len() - 1 { " ]}\n" } else { " ]},\n" }); + } + s.push_str(" ],\n"); + s.push_str(&format!(" \"dataset_head\": [{}],\n", join_jhex(&v.head))); + s.push_str(&format!(" \"dataset_last_index\": {mask},\n")); + s.push_str(&format!(" \"dataset_last\": {},\n", jhex(v.last))); + s.push_str(&format!( + " \"dataset_samples\": [{}]", + v.sample_idx + .iter() + .zip(v.sample_val.iter()) + .map(|(i, val)| format!("{{\"index\": {i}, \"value\": {}}}", jhex(*val))) + .collect::>() + .join(", ") + )); + if memhard { + s.push_str(&format!(",\n \"cache_head\": [{}],\n", join_jhex(&v.cache_head))); + s.push_str(&format!(" \"cache_last_line\": [{}],\n", join_jhex(&v.cache_last))); + s.push_str(&format!(" \"cache_fnv1a64\": {}", jhex64(v.cache_fnv))); + if v.has_hot { + s.push_str(&format!(",\n \"hot_head\": [{}],\n", join_jhex(&v.hot_head))); + s.push_str(&format!(" \"hot_last_line\": [{}],\n", join_jhex(&v.hot_last))); + s.push_str(&format!(" \"hot_fnv1a64\": {}", jhex64(v.hot_fnv))); + } + s.push('\n'); + } else { + s.push('\n'); + } + s.push_str("}\n"); + s +} + +/// A program pack: the files `--export-pack` writes, as (name, text). +pub struct Pack { + pub files: Vec<(String, String)>, + /// Binary files beside the texts: `leaves.bin` of a class v5 pack (empty for every other pack). + pub binaries: Vec<(String, Vec)>, + pub bases: Vec, + pub outs: Vec<[u64; 32]>, + pub vectors: PackVectors, +} + +impl Pack { + pub fn write_to(&self, dir: &std::path::Path) -> std::io::Result<()> { + std::fs::create_dir_all(dir)?; + for (name, text) in &self.files { + std::fs::write(dir.join(name), text)?; + } + for (name, bytes) in &self.binaries { + std::fs::write(dir.join(name), bytes)?; + } + Ok(()) + } +} + +/// Build the whole pack for an epoch: the three vector warps from the CPU interpreter, the self-test words, +/// and every source file. The `source` string says where the vectors came from. +pub fn export_pack(epoch: &Epoch, day: &str, source: &str) -> Pack { + let p = &epoch.program; + let ds: &DatasetSource = &epoch.dataset; + let mask = ds.mask; + let memhard = ds.memhard().map(|m| &m.params); + let bases = PACK_VECTOR_BASES.to_vec(); + let outs: Vec<[u64; 32]> = bases.iter().map(|&b| epoch.hash_warp(b)).collect(); + // the self-test words under the program's layout (era layout; linear for every other class) + let mut v = PackVectors { + head: (0..16).map(|i| epoch.dataset_word(i)).collect(), + last: epoch.dataset_word(mask), + sample_idx: sample_indices(mask), + ..Default::default() + }; + v.sample_val = v.sample_idx.iter().map(|&i| epoch.dataset_word(i)).collect(); + if let Some(m) = ds.memhard() { + let w = m.cache.words(); + v.cache_head = w[..16].to_vec(); + v.cache_last = w[w.len() - 16..].to_vec(); + v.cache_fnv = m.cache.fnv1a64(); + v.cache_log2_words = m.shape().cache_log2_words; + } + if let Some(h) = &ds.hot { + let w = h.words(); + v.has_hot = true; + v.hot_head = w[..16].to_vec(); + v.hot_last = w[w.len() - 16..].to_vec(); + v.hot_fnv = h.fnv1a64(); + } + let is_mh = memhard.is_some(); + let mut files = vec![ + ("program.json".to_string(), program_json(p, day, ds)), + ("vectors.json".to_string(), vectors_json(p, day, ds.log2_words, &bases, &outs, &v, mask, source, is_mh)), + ("kernel.cu".to_string(), cuda_kernel_at(p, memhard, ds.log2_words)), + ("kernel.cl".to_string(), opencl_kernel_at(p, memhard, ds.log2_words)), + ("program.h".to_string(), program_header(p, day, ds)), + ("vectors.h".to_string(), vectors_header(p, &bases, &outs, &v, mask, source, is_mh)), + ("program.metal".to_string(), metal_program(p, ds.log2_words, LoadSource::Stored)), + // Header-bound kernels (3 October 2026, bind.rs): new files, the seven above are unchanged. + ("program_bound.metal".to_string(), metal_program_bound(p, ds.log2_words)), + ("kernel_bound.cu".to_string(), cuda_kernel_bound_at(p, memhard, ds.log2_words)), + ("kernel_bound.cl".to_string(), opencl_kernel_bound_at(p, memhard, ds.log2_words)), + ]; + if let Some(mp) = memhard { + files.push(("memhard.h".to_string(), cuda_memhard_header(p, mp))); + files.push(("memhard.metal".to_string(), metal_memhard_for(p, mp))); + } + let binaries = match ds.leaves() { + Some(l) => vec![("leaves.bin".to_string(), l.bytes())], + None => Vec::new(), + }; + Pack { files, binaries, bases, outs, vectors: v } +} + +/// The dataset mode a pack was written in, from its program.json text (no JSON parser needed). +pub fn pack_mode_from_json(program_json: &str) -> DatasetMode { + if program_json.contains("\"dataset_mode\": \"memory-hard\"") { + DatasetMode::MemoryHard + } else { + DatasetMode::ClosedForm + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/generator.rs b/tools/attack/adv-accept-v5/igneum-pow/src/generator.rs new file mode 100644 index 000000000..46955d6c0 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/generator.rs @@ -0,0 +1,2699 @@ +//! The program generator: 64 integer instructions over 8 x u32 lane registers, run for 8 iterations. +//! +//! Generator version 2 (adopted 4 October 2026 from `docs/analysis/weak-program-census-2026-10-03.md`, +//! spec 01 sections 1.4.2, 1.4.3 and 1.4.6): +//! +//! * G1, exact load count: every program has exactly [`LOAD_SLOTS`] (16) `load` instructions, drawn first as a +//! uniform 16-subset of instruction slots 1..63 by a partial Fisher-Yates over the program stream. The other +//! 48 ops come from the ten non-load families with the weights of [`NONLOAD_WEIGHTS`] (sum 75). +//! * G2, fresh source: on a load slot the source register is drawn from `E`, the registers other than `dst` that +//! an earlier instruction of this program has written and that no later load has read. A load's address is then +//! a value produced in this iteration that no earlier load used, so no load of a hash repeats an earlier load's +//! address, across the iteration boundary included. If `E` is empty the source is drawn as on an ALU slot and +//! the acceptance rule of [`crate::accept`] rejects the program. +//! * R, acceptance: a candidate must pass [`crate::accept::check`]. A rejected candidate is replaced by the next +//! attempt, `seed_words_from_bytes(program_seed || k_le32)` for `k = 1, 2, ...` (attempt 0 is the bare seed), +//! so every node derives the same program from the same seed. +//! +//! The retired version 1 generator (op rolled per instruction with a 25 percent load weight, no acceptance) is +//! kept as [`generate_v1`] for the census tool and the lever measurements of `proto-metal/MEMHARD.md`. Its +//! programs are not the lottery hash and no pack or vector of version 1 is current. +//! +//! Era layout (5 October 2026, Counter ASIC 2.0 layers 4 and 8, `docs/plans/era-layout.md`; NOT the lottery hash, +//! behind [`LoadClass::era`]): [`EraParams`] drawn from the era seed `E_n` by [`era_draw`] (the load width, a stride +//! multiplier and rotation, the interleave of item words over the dataset), and per load site two more draws (a +//! window of the dataset: a half, a quarter or all of it, at a drawn offset). The load address is +//! [`crate::verify::load_index`]. An era class takes 12 draws per instruction, so its stream differs from version 2. +//! +//! Read-width experiment (5 October 2026, gate 1, `docs/plans/read-width.md`; NOT the lottery hash, behind +//! [`LoadClass`]): a program class whose `load` reads `W` bytes (4, 16 or 64: 1, 4 or 16 words, aligned to `W`) +//! and folds every word into `dst` (`verify::fold_words`), with the width fixed per class or drawn per load from +//! an era-fixed mix. The default class [`LoadClass::V2`] is the generator above, draw for draw and byte for byte; +//! every other class takes one extra draw per instruction (the width roll), so its program stream differs from +//! version 2 and its program id carries the class. +//! +//! Hot-table experiment (5 October 2026, Counter ASIC 2.0 layer 5, `docs/plans/hot-table.md`; NOT the lottery hash, +//! behind [`LoadClass::hot`]): `k` of the load slots read a second table `H` of `S` MiB derived from the epoch seed +//! ([`crate::memhard::HotTable`]) at `H[mulhi(src, words)]` with the plain one-word fold. The hot slots are the +//! first `k` drawn load slots after the scratch slots (a uniform `k`-subset, no extra draw), so a class with the +//! version 2 widths and no scratch takes the version 2 stream exactly ([`LoadClass::takes_width_roll`]). + +use crate::accept::{check, Reject}; +use crate::seed::{fnv1a64, program_rng, seed_words_from_bytes, SplitMix64}; + +/// Iterations of the instruction list per hash. +pub const ITERATIONS: usize = 8; +/// Instructions per program. +pub const INSTR_COUNT: usize = 64; +/// Lanes per verification unit (one SIMD group / warp). +pub const LANES: usize = 32; +/// The generator version written into every pack and program id. Version 1 programs never mix with these. +pub const GENERATOR_VERSION: u32 = 2; +/// Load instructions per program under version 2 (G1): 128 loads per hash, 4,096 items per 32-lane unit. +pub const LOAD_SLOTS: usize = 16; +/// Attempts before an implementation may treat the seed as a consensus fault (spec 01 section 1.4.6). At the +/// measured 5.14 percent rejection rate the chance of 32 consecutive rejections is below 2^-136. +pub const MAX_ATTEMPTS: u32 = 32; + +/// The attempt cap of class v4 sub-version 2 (AP-F8-2, 7 October 2026): rules (a') and (c') reject about two thirds of +/// candidates, so 32 attempts exhaust with probability about (2/3)^32, 2e-6 per epoch seed, one epoch no node could +/// draw every few decades at one epoch an hour (seen at chain-shaped seed igneum-f9/331672). At 256 attempts the +/// exhaustion probability is (2/3)^256, under 1e-45; the cost of a rejected attempt is one draw and the 64-unit check, +/// about 2 ms on one core, so the worst case is half a second. Keyed on the class v4 shape, so v2 and v3 keep 32. +pub const MAX_ATTEMPTS_V4: u32 = 256; + +/// The attempt cap of a class: [`MAX_ATTEMPTS_V4`] for the class v4 shape, [`MAX_ATTEMPTS`] otherwise. +pub fn max_attempts_for(class: &LoadClass) -> u32 { + if crate::accept::is_class_v4_shape(class) { MAX_ATTEMPTS_V4 } else { MAX_ATTEMPTS } +} +/// Domain tag of the program id. +pub const PROGRAM_ID_TAG: &[u8] = b"igneum-program/"; + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub enum Op { + Add, + Sub, + Mul, + MulHi, + Xor, + Or, + Rotl, + Rotr, + Mad, + Shfl, + Load, + /// Warp-coalesced load (lever b of the version 1 generator). Never emitted by version 2. + WLoad, + /// Scratch read-modify-write (read-width experiment, variant 5, 5 October 2026): a 16-byte slot of the lane's + /// own 32 KiB of the warp's 1 MiB scratch, read, folded into dst, rewritten. Never emitted by version 2. + Scratch, + /// Hot-table load (hot-table experiment, 5 October 2026): `dst = dst XOR H[mulhi(src, HOT_WORDS)]`, one word + /// of the epoch's `S` MiB table. Never emitted by version 2. + Hot, +} + +impl Op { + /// The name used in program.json, kernel comments and the op mix. + pub fn name(self) -> &'static str { + match self { + Op::Add => "add", + Op::Sub => "sub", + Op::Mul => "mul", + Op::MulHi => "mulhi", + Op::Xor => "xor", + Op::Or => "or", + Op::Rotl => "rotl", + Op::Rotr => "rotr", + Op::Mad => "mad", + Op::Shfl => "shfl", + Op::Load => "load", + Op::WLoad => "wload", + Op::Scratch => "scratch", + Op::Hot => "hot", + } + } + + pub fn from_name(s: &str) -> Option { + Some(match s { + "add" => Op::Add, + "sub" => Op::Sub, + "mul" => Op::Mul, + "mulhi" => Op::MulHi, + "xor" => Op::Xor, + "or" => Op::Or, + "rotl" => Op::Rotl, + "rotr" => Op::Rotr, + "mad" => Op::Mad, + "shfl" => Op::Shfl, + "load" => Op::Load, + "wload" => Op::WLoad, + "scratch" => Op::Scratch, + "hot" => Op::Hot, + _ => return None, + }) + } + + /// An injecting op: bijective in `dst` and bringing another register (or the dataset) in. The acceptance + /// rule's part (b) requires one such write per register. + pub fn injects(self) -> bool { + matches!(self, Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl | Op::Load | Op::WLoad | Op::Scratch | Op::Hot) + } + + /// A memory operation: the fresh-source rule, the acceptance tests and the load count treat the scratch + /// read-modify-write and the hot-table load as loads (each is one of the program's 128 memory operations). + pub fn is_load(self) -> bool { + matches!(self, Op::Load | Op::WLoad | Op::Scratch | Op::Hot) + } +} + +/// One instruction. Every field is drawn for every instruction whether the op uses it or not, so the +/// draw stream is identical for every op. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Instr { + pub op: Op, + /// Destination register 0..7. + pub dst: u8, + /// Source register 0..7, never equal to `dst`. + pub src: u8, + /// Second source (mad only). + pub src2: u8, + /// Add immediate A. + pub imm: u32, + /// Add immediate B. + pub imm2: u32, + /// rotl amount 1..31. + pub rot: u32, + /// Selector bit of r0 for add, 0..31. + pub bit: u8, + /// Shuffle xor mask: 1, 2, 4, 8 or 16. + pub mask: u8, + /// Words read by a `load`: 1 (the lottery hash, 4 bytes), 4 or 16 (the read-width experiment). 1 on every + /// other op. + pub width: u8, + /// Era layout, layer 8: the window shrink of this load site, 0..2 (the dataset, a half, a quarter). 0 on every + /// op of every other class. + pub win: u8, + /// Era layout, layer 8: which aligned window, below `2^win`. 0 on every op of every other class. + pub off: u8, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct Program { + /// A label for packs and logs: the seed string, or whatever the caller named a byte seed. + pub seed_string: String, + /// The program seed bytes (`program_seed` of spec 01 section 1.12): the UTF-8 of a string seed, the 32-byte + /// epoch seed on the chain. Attempt `k` of this seed is `seed_words_from_bytes(seed_bytes || k_le32)`. + pub seed_bytes: Vec, + /// The seed words of this attempt (what the program stream and the register init use). + pub seed: [u32; 8], + /// Generator version, [`GENERATOR_VERSION`] for every current program; 1 for the retired generator. + pub generator: u32, + /// Attempt index: 0 for the bare seed, `k` for the k-th re-derivation after rejections. + pub attempt: u32, + /// The load class: [`LoadClass::V2`] for the lottery hash, another for the read-width experiment. + pub class: LoadClass, + /// The era seed bytes a class v3 chain program was drawn under (`E_n` of spec 04 section 4.4, the devnet stand-in + /// of `docs/plans/era-layout.md` section 2), recorded in the pack so a worker can check it carries the era the + /// job names. `None` for every version 2 program and every string-seed pack. The placeholder [`V3_CLASS`] does + /// not read it; the era draw of the integration branch will. + pub era_bytes: Option>, + pub instrs: Vec, + /// The latency-shadow block (Counter ASIC 3.0 item 8): `class.shadow.instrs` ALU instructions, run + /// `class.shadow.reps` times at the end of every iteration. Empty for every class without a shadow. + pub shadow: Vec, +} + +/// The widths a `load` may read, in words: 4, 16 and 64 bytes. +pub const WIDTH_WORDS: [u8; 3] = [1, 4, 16]; + +/// The load class of a program (read-width experiment, 5 October 2026). `mix` holds the percent weights of the +/// three widths of [`WIDTH_WORDS`] (sum 100); `load_slots` the number of `load` instructions per program. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct LoadClass { + pub mix: [u8; 3], + pub load_slots: u8, + /// Variant 5: `Some(k)` gives the program a per-warp scratch (the kernels run persistent warps) and turns `k` + /// of the load slots into scratch read-modify-writes. `None` for every other class. + pub scratch: Option, + /// Variant 5: the scratch per warp in KiB (32 or 128; the whole working set of a card at full occupancy must + /// stay under 6 GB, coordinator's cap of 5 October 2026). 0 for every other class. + pub scratch_kb: u8, + /// Mixer cost multiplier `m` of the dataset item derivation (Counter ASIC 2.0, M16, decided 5 October 2026 for + /// class v3): every mixer application of spec 01 section 1.8.5 becomes `m` applications with distinct round + /// keys, the 8 dependent cache reads per item unchanged (`memhard::derive_items`). 1 for version 2, 4 for v3. + pub mixer_mult: u8, + /// Cache growth rule, option C (`memhard::growth_doublings`): the cache doubles when the dataset doubles. `false` + /// for version 2 (the cache is 2^26 words on every day), `true` for v3. + pub growth: bool, + /// Era layout (`docs/plans/era-layout.md`): `Some` turns on the strided, windowed load address and the + /// interleaved dataset mapping with the parameters drawn from the era seed. `None` for every other class. + pub era: Option, + /// Hot table (`docs/plans/hot-table.md`, measured 5 October 2026 and not adopted): `Some(HotClass { mb, k, added })` + /// turns `k` load slots into reads of an `mb` MiB epoch table. `None` for every other class, class v3 included. + pub hot: Option, + /// Counter ASIC 3.0 item 2 (`crate::derive`, `docs/plans/counter-asic-3-derivation.md`, 6 October 2026, a + /// prototype behind the class): instructions per round program of the per-day item-derivation program that + /// replaces the fixed mixer when non-zero (the mixer multiplier is then unused and 1). 0 for every other class. + pub derive_len: u16, + /// Latency-shadow program work (Counter ASIC 3.0 item 8, measured 6 October 2026 and not adopted): `Some` adds a + /// block of ALU instructions run `reps` times per iteration. `None` for every other class, class v3 included. + pub shadow: Option, + /// Class v5, proof of stored state and of following (`docs/design/class-v5-stored-state.md`, 7 October 2026): + /// the item derivation XORs the window's state leaf into every item before the first mixer (`crate::state`, + /// `memhard::derive_items_leaves`). The program draw does not read it. `false` for every other class. + pub state: bool, +} + +/// The parameters one era draws from its seed `E_n` (`docs/plans/era-layout.md` section 1.1, the proposed text of +/// spec 01 section 1.13.1). `Copy` so the class stays `Copy`; the eight words of the era stream's seed and the era +/// index are carried so a pack can say where the draw came from. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct EraParams { + /// `seed_words_from_bytes("igneum-era/" || E_n)` (`E_n` commits to the era index through the VDF input of spec + /// 04 section 4.4 step 2, so the index does not enter the draw). + pub words: [u32; 8], + /// The genesis-fixed set the width is drawn from, ascending, zero-padded (`[1, 0, 0]` pins 4 bytes: the + /// read-width decision of 5 October 2026, v2's 128 x 4 B stays). + pub allowed: [u8; 3], + /// Layer 4, item size: the words one load folds (1, 4 or 16), drawn from the allowed set. + pub width_words: u8, + /// Layer 4, stride: `y = rotl(x * stride_mul, stride_rot)`; the multiplier is odd, the rotation in 1..31. + pub stride_mul: u32, + pub stride_rot: u32, + /// Layer 4, interleave: the four ascending bit positions (0..15) of the word-within-item bits in the word + /// index; the first `log2(width_words)` are `0..`, so one aligned load stays inside one item. + pub pos: [u8; 4], +} + +/// Domain tag of the era stream seed. +pub const ERA_TAG: &[u8] = b"igneum-era/"; + +impl EraParams { + /// `seed_words_from_bytes("igneum-era/" || era_bytes)`. + pub fn stream_words(era_bytes: &[u8]) -> [u32; 8] { + let mut b = Vec::with_capacity(ERA_TAG.len() + era_bytes.len()); + b.extend_from_slice(ERA_TAG); + b.extend_from_slice(era_bytes); + seed_words_from_bytes(&b) + } + + /// The short label of the era in class names and pack lines: the first stream word as hex. + pub fn label(&self) -> String { + format!("{:08x}", self.words[0]) + } + + /// The era bytes of a test seed string: the 32 bytes (little-endian words) of `seed_words_from_bytes(s)`. + pub fn test_era_bytes(s: &str) -> [u8; 32] { + let w = seed_words_from_bytes(s.as_bytes()); + let mut out = [0u8; 32]; + for (i, x) in w.iter().enumerate() { + out[i * 4..i * 4 + 4].copy_from_slice(&x.to_le_bytes()); + } + out + } + + /// The dataset layout this era's loads and build use. + pub fn layout(&self) -> crate::memhard::Layout { + crate::memhard::Layout { pos: self.pos } + } + + /// The allowed widths as a slice (the non-zero entries). + pub fn allowed_set(&self) -> Vec { + self.allowed.iter().copied().filter(|&w| w != 0).collect() + } + + /// The bytes that enter the program id after `"era/"`: the allowed set, width, multiplier, rotation, positions. + pub fn id_bytes(&self) -> Vec { + let mut b = Vec::with_capacity(3 + 1 + 4 + 4 + 4); + b.extend_from_slice(&self.allowed); + b.push(self.width_words); + b.extend_from_slice(&self.stride_mul.to_le_bytes()); + b.extend_from_slice(&self.stride_rot.to_le_bytes()); + b.extend_from_slice(&self.pos); + b + } +} + +/// The era draw (`docs/plans/era-layout.md` section 1.1): seven draws from one SplitMix64 stream seeded with words 0 +/// and 1 of [`EraParams::stream_words`], in this order: the width from `allowed` (ascending, a non-empty subset of +/// [`WIDTH_WORDS`]; one element pins it, the draw is still consumed), the odd stride multiplier, the stride rotation +/// in 1..31, then four draws for the interleave (a partial Fisher-Yates over the candidate positions `log2(W)..15`, +/// `4 - log2(W)` of them used, the rest consumed). +pub fn era_draw(era_bytes: &[u8], allowed: &[u8]) -> EraParams { + assert!(!allowed.is_empty() && allowed.len() <= 3, "the allowed width set has 1 to 3 entries"); + for (i, &w) in allowed.iter().enumerate() { + assert!(WIDTH_WORDS.contains(&w), "allowed width {w} is not 1, 4 or 16 words"); + assert!(i == 0 || allowed[i - 1] < w, "the allowed width set is ascending"); + } + let words = EraParams::stream_words(era_bytes); + let mut s = SplitMix64::new(words[0] as u64 | ((words[1] as u64) << 32)); + let width_words = allowed[s.below(allowed.len() as u64) as usize]; + let stride_mul = (s.next() as u32) | 1; + let stride_rot = 1 + s.below(31) as u32; + let b = width_words.trailing_zeros() as usize; // 0, 2 or 4 + let free = 4 - b; + let mut c: Vec = (b as u8..16).collect(); + let mut r = [0u64; 4]; + for x in r.iter_mut() { + *x = s.next(); + } + for i in 0..free { + let n = c.len() - i; + let j = i + (r[i] % n as u64) as usize; + c.swap(i, j); + } + // Draws 8 and 9, consumed and not used (spec 01 section 1.13.1 for `epoch_len`; `docs/design/latency-ladder.md` + // section 2 for the latency ladder): both parameters are set by miner signal, and consuming their slots here means a + // later use of either changes no other draw. Nothing below reads them, so every value drawn above is what it was. + let _epoch_len_draw = s.next(); + let _latency_ladder_draw = s.next(); + let mut chosen: Vec = c[..free].to_vec(); + chosen.sort_unstable(); + let mut pos = [0u8; 4]; + for i in 0..b { + pos[i] = i as u8; + } + for (i, p) in chosen.iter().enumerate() { + pos[b + i] = *p; + } + let mut al = [0u8; 3]; + al[..allowed.len()].copy_from_slice(allowed); + EraParams { words, allowed: al, width_words, stride_mul, stride_rot, pos } +} + +/// The hot table of a class: `mb` MiB (32, 64 or 96 in the experiment) and `k` hot slots. Two forms: `replaced` +/// (`added = false`): `k` of the 16 load slots read the table, 16 - k dataset loads; `added` (`added = true`, +/// coordinator's form of 5 October 2026 against the on-die-cache recompute chip): the program has 16 + k load slots, +/// the `k` hot ones drawn among them, so the 16 dataset loads and the 4,096-item verifier bound are unchanged and the +/// hot loads are extra work (a cache hit on a GPU, SRAM and a read on a chip). +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct HotClass { + pub mb: u8, + pub k: u8, + pub added: bool, +} + +/// Program work in the latency shadow (Counter ASIC 3.0 item 8, 6 October 2026, `docs/analysis/latency-shadow-2026-10-06.md`; +/// NOT the lottery hash, a class v4 candidate's knob): a shadow block of `instrs` ALU instructions (the ten non-load +/// families at the weights of [`NONLOAD_WEIGHTS`], the same nine draws per instruction as the program's, drawn from the +/// program stream AFTER the 64 base instructions, so the base program, its attempt and its acceptance verdict are those +/// of the class without the shadow, draw for draw) executed `reps` times at the end of every iteration, after instruction +/// 63 and before the next iteration samples `sel`. The 16 loads per program, the fresh-source rule and the acceptance +/// rule (which interprets the base program only) are untouched; the shadow adds `8 x instrs x reps` ALU instructions per +/// hash and no load. `None` on every other class: version 2 and class v3 draw nothing and emit nothing. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct ShadowClass { + /// Instructions in the shadow block (1..=4096). + pub instrs: u16, + /// Times the block runs per iteration (1..=1024). + pub reps: u16, +} + +impl ShadowClass { + /// Shadow instructions per hash: `ITERATIONS x instrs x reps`. + pub fn instrs_per_hash(&self) -> usize { + ITERATIONS * self.instrs as usize * self.reps as usize + } +} + +/// Scratch geometry (variant 5): 16-byte slots, lane-major, 32 lanes per warp; `scratch_kb` KiB per warp gives +/// `scratch_kb x 2` slots per lane (32 KiB: 64 slots, 128 KiB: 256 slots). +pub const SCRATCH_SLOT_BYTES: usize = 16; + +impl LoadClass { + /// Slots per lane of the scratch (0 without one). + pub fn scratch_slots_per_lane(&self) -> usize { + self.scratch_kb as usize * 1024 / LANES / SCRATCH_SLOT_BYTES + } + pub fn scratch_slot_mask(&self) -> u32 { + self.scratch_slots_per_lane().saturating_sub(1) as u32 + } + pub fn scratch_words_per_lane(&self) -> usize { + self.scratch_slots_per_lane() * 4 + } + pub fn scratch_bytes_per_warp(&self) -> usize { + self.scratch_kb as usize * 1024 + } +} + +impl LoadClass { + /// Generator version 2 as adopted on 4 October 2026: 16 loads of one word. The lottery hash. + pub const V2: LoadClass = + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 1, growth: false, era: None, hot: None, derive_len: 0, shadow: None, state: false }; + + /// The construction decided for program class v3 on 5 October 2026 (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): + /// version 2 loads (16 slots of one word, no scratch, no width roll, so the program stream is version 2's), the + /// mixer applied 4 times per round, and the cache growth rule. Name "mx4". + pub const MX4: LoadClass = + LoadClass { mix: [100, 0, 0], load_slots: LOAD_SLOTS as u8, scratch: None, scratch_kb: 0, mixer_mult: 4, growth: true, era: None, hot: None, derive_len: 0, shadow: None, state: false }; + + /// The era class over `base` (`docs/plans/era-layout.md`): the parameters drawn by [`era_draw`]; when `allowed` + /// has more than one width the drawn width becomes the class mix (every load that width), otherwise the base + /// class's mix stands (the width rule of the read-width branch, pinned) and the era's `width_words` is the widest + /// width that mix can draw. + pub fn era(base: LoadClass, era_bytes: &[u8], allowed: &[u8]) -> LoadClass { + let mut e = era_draw(era_bytes, allowed); + let mut c = base; + if allowed.len() > 1 { + let i = WIDTH_WORDS.iter().position(|&w| w == e.width_words).unwrap(); + c.mix = [0, 0, 0]; + c.mix[i] = 100; + } else { + let widest = (0..3).rev().find(|&i| c.mix[i] > 0).map(|i| WIDTH_WORDS[i]).unwrap_or(1); + if widest != e.width_words { + // the interleave must keep the widest load inside one item: redraw the positions for that width + e = era_draw(era_bytes, &[widest]); + } + } + c.era = Some(e); + c + } + + /// The dataset layout of this class ([`crate::memhard::Layout::LINEAR`] without an era). + pub fn layout(&self) -> crate::memhard::Layout { + self.era.map(|e| e.layout()).unwrap_or(crate::memhard::Layout::LINEAR) + } + + /// The x8 candidate beside [`LoadClass::MX4`] (coordinator's rule of 5 October 2026, 21:30 UTC: x8 enters v3 if + /// the per-warp verify stays under 10 ms on one Mac core and the daily 1 GiB build under 1 s on every card): + /// the same loads and growth rule, the mixer applied 8 times per round. Name "mx8". + pub const MX8: LoadClass = + LoadClass { mixer_mult: 8, ..LoadClass::MX4 }; + + /// Counter ASIC 3.0 item 2 (6 October 2026, `docs/plans/counter-asic-3-derivation.md`): version 2 loads and the + /// growth rule of class v3, with the per-day derivation program of `crate::derive` (736 instructions per round + /// program, the x8-equivalent operation count) in place of the mixer. Name "dr736". A prototype class; not v3. + pub const DR736: LoadClass = + LoadClass { mixer_mult: 1, derive_len: crate::derive::DERIVE_LEN_X8 as u16, ..LoadClass::MX4 }; + + /// This class with a derivation program of `len` instructions per round program (0: the fixed mixer); the + /// mixer multiplier is set to 1 under a program, since no mixer is applied. + pub fn with_derive(self, len: u16) -> LoadClass { + assert!(len == 0 || (32..=4096).contains(&len), "derivation program length must be 0 or 32..=4096"); + LoadClass { derive_len: len, mixer_mult: if len != 0 { 1 } else { self.mixer_mult }, ..self } + } + + /// Whether the item derivation is the per-day program (Counter ASIC 3.0 item 2). + pub fn is_derived(&self) -> bool { + self.derive_len != 0 + } + + /// A fixed width (1, 4 or 16 words) with `load_slots` loads per program. + pub fn fixed(width_words: u8, load_slots: u8) -> LoadClass { + let mut mix = [0u8; 3]; + let i = WIDTH_WORDS.iter().position(|&w| w == width_words).expect("width must be 1, 4 or 16 words"); + mix[i] = 100; + LoadClass { mix, load_slots, ..LoadClass::V2 } + } + + /// Per-load width drawn from `mix` (percent for 4, 16, 64 bytes), 16 loads per program. + pub fn mixed(mix: [u8; 3]) -> LoadClass { + assert_eq!(mix.iter().map(|&m| m as u32).sum::(), 100, "the mix must sum to 100"); + LoadClass { mix, ..LoadClass::V2 } + } + + /// Variant 5: version 2 widths, 16 memory operations of which `k` are scratch read-modify-writes into a + /// scratch of `kb` KiB per warp (a power of two, 1 to 128: at least one slot per lane, under the 6 GB cap). + pub fn scratch(k: u8, kb: u8) -> LoadClass { + assert!(k as usize <= LOAD_SLOTS); + assert!(kb.is_power_of_two() && kb <= 128, "scratch per warp must be a power of two up to 128 KiB"); + LoadClass { scratch: Some(k), scratch_kb: kb, ..LoadClass::V2 } + } + + /// Hot table, replaced form: version 2 widths, 16 load slots of which `k` read an `mb` MiB epoch table (`hot64k4`). + pub fn hot(mb: u8, k: u8) -> LoadClass { + LoadClass::V2.with_hot(mb, k) + } + + /// Hot table, added form: version 2 widths, 16 + `k` load slots of which `k` read the table (`hot64k4a`), the 16 + /// dataset loads unchanged. + pub fn hot_added(mb: u8, k: u8) -> LoadClass { + LoadClass::V2.with_hot_added(mb, k) + } + + /// This class with a hot table in the replaced form (composes with a width mix or a scratch: the hot slots are + /// drawn after the scratch slots, and `scratch + hot` must fit the slot count). + pub fn with_hot(mut self, mb: u8, k: u8) -> LoadClass { + assert!(mb >= 1, "a hot table needs at least 1 MiB"); + assert!(self.scratch_slots() + k as usize <= self.load_slots as usize, "scratch and hot slots exceed the load slots"); + self.hot = Some(HotClass { mb, k, added: false }); + self + } + + /// This class with a hot table in the added form: `k` load slots are added to the class's and read the table. + pub fn with_hot_added(mut self, mb: u8, k: u8) -> LoadClass { + assert!(mb >= 1, "a hot table needs at least 1 MiB"); + assert!((self.load_slots as usize) + (k as usize) < INSTR_COUNT, "added hot slots exceed the instruction count"); + self.load_slots += k; + self.hot = Some(HotClass { mb, k, added: true }); + self + } + + /// Hot slots per program (0 without a hot table). + pub fn hot_slots(&self) -> usize { + self.hot.map(|h| h.k as usize).unwrap_or(0) + } + + /// Hot slots that were added to the slot count (0 for the replaced form and without a hot table). + pub fn hot_added_slots(&self) -> usize { + self.hot.map(|h| if h.added { h.k as usize } else { 0 }).unwrap_or(0) + } + + /// Load slots that read the dataset: the slot count less the scratch and hot slots. + pub fn dataset_slots(&self) -> usize { + self.load_slots as usize - self.scratch_slots() - self.hot_slots() + } + + /// This class with the mixer multiplier `m` (1, 2, 4, 8 or 16) and the cache growth rule on or off. + /// The class with a latency-shadow block of `instrs` instructions run `reps` times per iteration ("mx8+sh256x13"). + pub fn with_shadow(self, instrs: u16, reps: u16) -> LoadClass { + assert!((1..=4096).contains(&instrs) && (1..=1024).contains(&reps), "shadow block: 1..=4096 instructions, 1..=1024 reps"); + LoadClass { shadow: Some(ShadowClass { instrs, reps }), ..self } + } + + /// This class with another class's era draw (tests: a rung's class composed with the chain's era). + pub fn with_era_of(self, other: &LoadClass) -> LoadClass { + LoadClass { era: other.era, ..self } + } + + /// The class with the state leaves of class v5 folded into every item ("mx8+sh256x27+state"). + pub fn with_state(self) -> LoadClass { + LoadClass { state: true, ..self } + } + + /// Shadow instructions per hash (0 without a shadow). + pub fn shadow_instrs_per_hash(&self) -> usize { + self.shadow.map(|s| s.instrs_per_hash()).unwrap_or(0) + } + + pub fn with_mixer(self, mixer_mult: u8, growth: bool) -> LoadClass { + assert!(mixer_mult >= 1 && mixer_mult <= 16 && mixer_mult.is_power_of_two(), "mixer multiplier must be 1, 2, 4, 8 or 16"); + LoadClass { mixer_mult, growth, ..self } + } + + /// The mixer multiplier as the item derivation uses it. + pub fn mixer_mult(&self) -> u32 { + self.mixer_mult as u32 + } + + /// Whether the loads of this class are version 2's: 16 one-word loads, no scratch. Such a class takes no width + /// roll, so its program stream is the version 2 stream draw for draw (the mixer and the cache are properties of + /// the dataset, not of the program). + pub fn v2_loads(&self) -> bool { + self.mix == [100, 0, 0] && self.load_slots as usize == LOAD_SLOTS + self.hot_added_slots() && self.scratch.is_none() + } + + /// Whether every instruction takes the tenth draw (the width roll): every class whose loads are not version 2's. + pub fn takes_width_roll(&self) -> bool { + !self.v2_loads() + } + + /// Parse "p4,p16,p64" or one of the names of [`LoadClass::name`] ("scr4k32": 4 scratch ops, 32 KiB per warp; + /// "mx4": the v3 construction; a trailing "m" and "g" set the mixer multiplier and the growth rule on any + /// load class, "w16m4g" for example). + pub fn parse(s: &str) -> Option { + // "+state": the state leaves of class v5 over any class (the suffix is outermost) + if let Some(base) = s.strip_suffix("+state") { + return Some(LoadClass::parse(base)?.with_state()); + } + // "+shx": the latency-shadow block over any class (Counter ASIC 3.0 item 8) + if let Some((base, sh)) = s.rsplit_once("+sh") { + let (instrs, reps) = sh.split_once('x')?; + let (instrs, reps): (u16, u16) = (instrs.parse().ok()?, reps.parse().ok()?); + if !(1..=4096).contains(&instrs) || !(1..=1024).contains(&reps) { + return None; + } + return Some(LoadClass::parse(base)?.with_shadow(instrs, reps)); + } + if s == "mx4" { + return Some(LoadClass::MX4); + } + if s == "mx8" { + return Some(LoadClass::MX8); + } + // Counter ASIC 3.0 item 2: "dr" is the derivation class on the v3 loads and growth rule + if let Some(digits) = s.strip_prefix("dr") { + if !digits.is_empty() && digits.bytes().all(|b| b.is_ascii_digit()) { + let len: u16 = digits.parse().ok()?; + if len == 0 || !(32..=4096).contains(&len) { + return None; + } + return Some(LoadClass::MX4.with_derive(len)); + } + } + // the mixer suffix: "...m" then an optional "g" + let (s, growth) = match s.strip_suffix('g') { + Some(base) if base.rsplit_once('m').map(|(_, d)| !d.is_empty() && d.bytes().all(|b| b.is_ascii_digit())).unwrap_or(false) => (base, true), + _ => (s, false), + }; + if let Some((base, digits)) = s.rsplit_once('m') { + if !digits.is_empty() && digits.bytes().all(|b| b.is_ascii_digit()) && !base.is_empty() && !base.ends_with(',') { + let mult: u8 = digits.parse().ok()?; + if mult == 0 || mult > 16 || !mult.is_power_of_two() { + return None; + } + return Some(LoadClass::parse_loads(base)?.with_mixer(mult, growth)); + } + } + if growth { + return None; + } + LoadClass::parse_loads(s) + } + + /// The load part of a class name (no mixer suffix). + fn parse_loads(s: &str) -> Option { + // "+hotk[a]" composes a hot table with any class; "hotk[a]" alone is the version 2 base; + // the "a" suffix is the added form (k slots added to the class's), without it the replaced form + if let Some((base, hot)) = s.split_once("+hot") { + let (added, hot) = match hot.strip_suffix('a') { Some(h) => (true, h), None => (false, hot) }; + let (mb, k) = hot.split_once('k')?; + let (mb, k): (u8, u8) = (mb.parse().ok()?, k.parse().ok()?); + let c = LoadClass::parse_loads(base)?; + if mb == 0 || k == 0 { + return None; + } + if added { + if c.load_slots as usize + k as usize >= INSTR_COUNT { + return None; + } + return Some(c.with_hot_added(mb, k)); + } + if c.scratch_slots() + k as usize > c.load_slots as usize { + return None; + } + return Some(c.with_hot(mb, k)); + } + if let Some(rest) = s.strip_prefix("hot") { + let (added, rest) = match rest.strip_suffix('a') { Some(h) => (true, h), None => (false, rest) }; + let (mb, k) = rest.split_once('k')?; + let (mb, k): (u8, u8) = (mb.parse().ok()?, k.parse().ok()?); + if mb == 0 || k == 0 || k as usize > LOAD_SLOTS { + return None; + } + return Some(if added { LoadClass::hot_added(mb, k) } else { LoadClass::hot(mb, k) }); + } + if let Some(rest) = s.strip_prefix("scr") { + let (k, kb) = rest.split_once('k')?; + let k: u8 = k.parse().ok()?; + let kb: u8 = kb.parse().ok()?; + if k as usize > LOAD_SLOTS || !kb.is_power_of_two() || kb > 128 { + return None; + } + return Some(LoadClass::scratch(k, kb)); + } + let (mix_s, slots) = match s.split_once("x") { + Some((m, n)) if !m.contains(',') => (m, n.parse::().ok()?), + _ => (s, LOAD_SLOTS as u8), + }; + let mix: [u8; 3] = match mix_s { + "v2" => return Some(LoadClass::V2), + "w4" => [100, 0, 0], + "w16" => [0, 100, 0], + "w64" => [0, 0, 100], + m => { + let v: Vec = m.split(',').map(|x| x.trim().parse::().ok()).collect::>>()?; + if v.len() != 3 || v.iter().map(|&x| x as u32).sum::() != 100 { + return None; + } + [v[0], v[1], v[2]] + } + }; + if slots == 0 || slots as usize >= INSTR_COUNT { + return None; + } + Some(LoadClass { mix, load_slots: slots, ..LoadClass::V2 }) + } + + /// Scratch read-modify-writes per program (0 without a scratch). + pub fn scratch_slots(&self) -> usize { + self.scratch.unwrap_or(0) as usize + } + + pub fn is_v2(&self) -> bool { + *self == LoadClass::V2 + } + + /// "v2", "w4", "w16", "w64", "w64x4", "mix50-35-15", "mix25-50-25x8", "scr4k32"; "mx4" for the v3 construction, "mx8" for its x8 candidate; + /// any other mixer setting appends "m" and, with the growth rule, "g" ("v2m2", "w16m4g"). + /// An era class is the base name with "-era" appended ("w4-era401998a5", "mx4-era..."). + /// A hot class appends "hotk[a]" ("hot64k4", "scr4k32+hot64k4a"; measured and not adopted). + pub fn name(&self) -> String { + if self.state { + // "+state": class v5's leaves are a suffix on any class, outermost + return format!("{}+state", LoadClass { state: false, ..*self }.name()); + } + if let Some(sh) = self.shadow { + // "+shx": the shadow block is a suffix on any class ("mx8+sh256x13") + return format!("{}+sh{}x{}", LoadClass { shadow: None, ..*self }.name(), sh.instrs, sh.reps); + } + if let Some(e) = self.era { + let base = LoadClass { era: None, ..*self }; + let base_name = if base.is_v2() { "w4".to_string() } else { base.name() }; + return format!("{base_name}-era{}", e.label()); + } + let base = LoadClass { hot: None, load_slots: self.load_slots - self.hot_added_slots() as u8, ..*self }.base_name(); + match self.hot { + None => base, + Some(h) if base == "v2" => format!("hot{}k{}{}", h.mb, h.k, if h.added { "a" } else { "" }), + Some(h) => format!("{base}+hot{}k{}{}", h.mb, h.k, if h.added { "a" } else { "" }), + } + } + + /// The name without the hot table. + fn base_name(&self) -> String { + if self.is_v2() { + return "v2".to_string(); + } + if *self == LoadClass::MX4 { + return "mx4".to_string(); + } + if *self == LoadClass::MX8 { + return "mx8".to_string(); + } + if self.derive_len != 0 { + // the derivation class: "dr" on the v3 loads; any other base keeps its name with the suffix + let base = LoadClass { derive_len: 0, mixer_mult: 1, ..*self }.base_name(); + return if base == "v2m1g" || base == "v2" { format!("dr{}", self.derive_len) } else { format!("{base}dr{}", self.derive_len) }; + } + let loads = LoadClass { mixer_mult: 1, growth: false, ..*self }; + let base = if loads.is_v2() { + "v2".to_string() + } else if let Some(k) = self.scratch { + format!("scr{k}k{}", self.scratch_kb) + } else { + let base = match self.mix { + [100, 0, 0] => "w4".to_string(), + [0, 100, 0] => "w16".to_string(), + [0, 0, 100] => "w64".to_string(), + [a, b, c] => format!("mix{a}-{b}-{c}"), + }; + if self.load_slots as usize == LOAD_SLOTS { + base + } else { + format!("{base}x{}", self.load_slots) + } + }; + let mut s = base; + if self.mixer_mult != 1 || self.growth { + s.push_str(&format!("m{}", self.mixer_mult)); + } + if self.growth { + s.push('g'); + } + s + } + + /// The width in words of a load whose width roll (0..99) is `roll`: the first entry of the mix whose cumulative + /// weight exceeds the roll. + pub fn width_for_roll(&self, roll: u64) -> u8 { + let mut acc = 0u64; + for (i, &m) in self.mix.iter().enumerate() { + acc += m as u64; + if roll < acc { + return WIDTH_WORDS[i]; + } + } + WIDTH_WORDS[2] + } + + /// Expected dataset bytes read per hash: dataset loads per hash times the mean width (scratch traffic apart). + pub fn expected_bytes_per_hash(&self) -> f64 { + let mean = self.mix.iter().zip(WIDTH_WORDS.iter()).map(|(&m, &w)| m as f64 / 100.0 * w as f64 * 4.0).sum::(); + (self.load_slots as usize - self.scratch_slots()) as f64 * ITERATIONS as f64 * mean + } +} + +impl Default for LoadClass { + fn default() -> Self { + LoadClass::V2 + } +} + +/// Generator version of a class v3 program (spec 01 section 1.4.6: `program_id(3, seed, attempt)`). +pub const GENERATOR_VERSION_V3: u32 = 3; + +/// Generator version of a class v4 program (Counter ASIC 3.0, 6 October 2026, PROPOSED: `program_id(4, seed, attempt)`). +pub const GENERATOR_VERSION_V4: u32 = 4; + +/// Generator version of a class v5 program (proof of stored state and of following, 7 October 2026, PROPOSED: +/// `program_id(5, seed, attempt)`; `docs/design/class-v5-stored-state.md`). +pub const GENERATOR_VERSION_V5: u32 = 5; + +/// The program class of an epoch (Counter ASIC 2.0, 5 October 2026, `docs/plans/counter-asic-2-rollout.md`): one +/// height switch in the node, `program_class_v3_activation_daa`, rounded up to an epoch boundary, decides which +/// class an epoch's program is drawn from. V2 is the lottery hash as adopted on 4 October 2026, byte for byte. +/// V3 is generator version 3: its program id carries `generator = 3` and its load class is [`V3_CLASS`]. +/// V4 (Counter ASIC 3.0, 6 October 2026, the candidate `mx8+sh256x27` behind `program_class_v4_activation_daa`) is +/// generator version 4: its program id carries `generator = 4` and its load class is [`V4_CLASS`]. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Default)] +pub enum ProgramClass { + #[default] + V2, + V3, + V4, + /// Class v5 (`docs/design/class-v5-stored-state.md`, behind `program_class_v5_activation_daa`): class v4's program + /// over a dataset whose every item is keyed by the window's execution state ([`V5_CLASS`]), generator 5. + V5, +} + +/// The load class of program class v3, decided 5 October 2026 (Counter ASIC 2.0, `docs/plans/counter-asic-2-status.md` +/// "22:00 decided", `docs/plans/mixer-x4.md`): [`LoadClass::MX4`], version 2 loads (the width stays 4 bytes, the +/// per-load mix and the scratch share are out), the mixer applied 4 times per round and the cache growth rule. The +/// placeholder of the seam (w16) is replaced here; nothing else in the seam names the class. +/// Composed on 5 October 2026 (branch ca2-era): the era layout of `docs/plans/era-layout.md` is drawn inside this class by +/// [`generate_from_seed_bytes_program_class`] (`LoadClass::era(V3_CLASS, era, &V3_ALLOWED)`); here `era` is `None`. +pub const V3_CLASS: LoadClass = LoadClass { era: None, hot: None, ..LoadClass::MX8 }; + +/// The load class of program class v4 (Counter ASIC 3.0 item 8, `docs/analysis/latency-shadow-2026-10-06.md`, the +/// candidate of 6 October 2026, gated by `docs/plans/counter-asic-3-node.md`): class v3 plus the latency-shadow block +/// of 256 ALU instructions run 27 times per iteration ("mx8+sh256x27", 55,296 shadow instructions per hash). The +/// base program, the 16 loads, the item construction, the cache growth rule and the era draw are class v3's, draw +/// for draw, so a v4 epoch's day cache and dataset are the v3 day's. Composed with the era exactly as V3 is. +pub const V4_CLASS: LoadClass = LoadClass { shadow: Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps: V4_SHADOW_REPS }), ..V3_CLASS }; + +/// The load class of program class v5 (`docs/design/class-v5-stored-state.md`, 7 October 2026): class v4 with the +/// state leaves of the window's reference block folded into every item of the dataset ("mx8+sh256x27+state"). The +/// program draw, the shadow block, the era draw and the ladder rung are class v4's, draw for draw; only the item +/// derivation and the program id change. +pub const V5_CLASS: LoadClass = LoadClass { state: true, ..V4_CLASS }; + +/// The shadow block size of class v4 at every rung of the latency ladder (`docs/design/latency-ladder.md`): 256 +/// instructions. The ladder moves the pass count alone. +pub const V4_SHADOW_INSTRS: u16 = 256; + +/// The shadow passes of class v4 at rung 0 of the latency ladder: 27 (`mx8+sh256x27`, about 102,100 counted ops). +pub const V4_SHADOW_REPS: u16 = 27; + +/// Class v4 at a rung of the latency ladder (`docs/design/latency-ladder.md` section 2): [`V4_CLASS`] with the +/// 256-instruction shadow block run `reps` times per iteration. `reps` 0 means the class's own count, so +/// `v4_class_at(0) == v4_class_at(27) == V4_CLASS` and a caller that knows no rung changes nothing. +pub fn v4_class_at(reps: u16) -> LoadClass { + if reps == 0 { + V4_CLASS + } else { + LoadClass { shadow: Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps }), ..V3_CLASS } + } +} + +/// Class v5 at a rung of the latency ladder: [`v4_class_at`] with the state leaves (`v5_class_at(0) == V5_CLASS`). +pub fn v5_class_at(reps: u16) -> LoadClass { + v4_class_at(reps).with_state() +} + +/// The shadow passes of a class v4 load class at any rung of the ladder, the era draw set aside (`Some(27)` for +/// [`V4_CLASS`] itself); `None` for every other class, a measurement class with another block size included. +pub fn v4_rung_reps(class: &LoadClass) -> Option { + let base = LoadClass { era: None, ..*class }; + if base.state { + return None; + } + match base.shadow { + Some(ShadowClass { instrs: V4_SHADOW_INSTRS, reps }) if LoadClass { shadow: None, ..base } == V3_CLASS => Some(reps), + _ => None, + } +} + +/// Counted integer ops per hash of class v4 at `reps` shadow passes, the convention of +/// `docs/analysis/latency-shadow-2026-10-06.md` section 1: 930 for the base program and its loads, 1.83 ops per +/// shadow instruction (137 / 75 over the non-load weights), 8 iterations x 256 instructions x `reps` shadow +/// instructions. Approximate by construction; the label of a rung, never a consensus value. +pub fn v4_counted_ops(reps: u16) -> u64 { + 930 + (ITERATIONS as u64 * V4_SHADOW_INSTRS as u64 * reps as u64 * 183).div_ceil(100) +} + +/// The width set class v3's era draw chooses from: 4 bytes only (the read-width decision of 5 October 2026; the +/// draw is consumed, so widening the set at genesis keeps the derivation). +pub const V3_ALLOWED: [u8; 1] = [1]; + +/// The program of an era class (generator version 3): `base` with the era parameters drawn from `era_bytes` +/// over the width set `allowed`, generator 3 stamped and the era bytes recorded (`docs/plans/era-layout.md`). +/// The chain's path is this with `base = V3_CLASS` and `allowed = V3_ALLOWED`. +pub fn generate_era(seed_string: &str, seed_bytes: &[u8], base: LoadClass, era_bytes: &[u8], allowed: &[u8]) -> Program { + generate_era_generator(seed_string, seed_bytes, base, era_bytes, allowed, era_generator_of(&base)) +} + +/// The generator version an era program over `base` is stamped with: 4 when the base (minus any era) is +/// [`V4_CLASS`], else 3. Counter ASIC 3.0, 6 October 2026: the seven gate packs exported through the CLI's `--era` +/// path on `mx8+sh256x27` were stamped generator 3 and so carried the v3 control's program id (`program_id(3, seed, +/// attempt)` is class-independent inside a generator version); a class v4 program is generator 4 wherever it is made. +pub fn era_generator_of(base: &LoadClass) -> u32 { + match ProgramClass::of_load_class(base) { + Some(ProgramClass::V4) => GENERATOR_VERSION_V4, + Some(ProgramClass::V5) => GENERATOR_VERSION_V5, + _ if base.state => GENERATOR_VERSION_V5, + _ => GENERATOR_VERSION_V3, + } +} + +/// The shadow passes of a class v5 load class at any rung (the state flag set aside); `None` for every other class. +pub fn v5_rung_reps(class: &LoadClass) -> Option { + if !class.state { + return None; + } + v4_rung_reps(&LoadClass { state: false, ..*class }) +} + +/// [`generate_era`] with the generator version stamped by the caller: 3 for class v3 over [`V3_CLASS`], 4 for +/// class v4 over [`V4_CLASS`] (Counter ASIC 3.0; the shadow block of the base rides through `LoadClass::era`). +pub fn generate_era_generator(seed_string: &str, seed_bytes: &[u8], base: LoadClass, era_bytes: &[u8], allowed: &[u8], generator: u32) -> Program { + let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, LoadClass::era(base, era_bytes, allowed)); + p.generator = generator; + p.era_bytes = Some(era_bytes.to_vec()); + p +} + +impl ProgramClass { + /// The load class this program class draws from. + pub fn load_class(&self) -> LoadClass { + match self { + ProgramClass::V2 => LoadClass::V2, + ProgramClass::V3 => V3_CLASS, + ProgramClass::V4 => V4_CLASS, + ProgramClass::V5 => V5_CLASS, + } + } + + /// The generator version written into every pack and program id of this class. + pub fn generator_version(&self) -> u32 { + match self { + ProgramClass::V2 => GENERATOR_VERSION, + ProgramClass::V3 => GENERATOR_VERSION_V3, + ProgramClass::V4 => GENERATOR_VERSION_V4, + ProgramClass::V5 => GENERATOR_VERSION_V5, + } + } + + /// Whether the class's dataset is keyed by the window's execution state (class v5). + pub fn has_state(&self) -> bool { + *self == ProgramClass::V5 + } + + /// The class of a generator version: 2, 3 and 4 are the three classes, anything else is no class this crate runs. + pub fn from_generator(generator: u32) -> Option { + match generator { + GENERATOR_VERSION => Some(ProgramClass::V2), + GENERATOR_VERSION_V3 => Some(ProgramClass::V3), + GENERATOR_VERSION_V4 => Some(ProgramClass::V4), + GENERATOR_VERSION_V5 => Some(ProgramClass::V5), + _ => None, + } + } + + /// "v2", "v3" or "v4": the `IGNEUM_PROGRAM_CLASS` string of a pack and the `class=` token of a job line. + pub fn name(&self) -> &'static str { + match self { + ProgramClass::V2 => "v2", + ProgramClass::V3 => "v3", + ProgramClass::V4 => "v4", + ProgramClass::V5 => "v5", + } + } + + pub fn parse(s: &str) -> Option { + match s.trim() { + "v2" => Some(ProgramClass::V2), + "v3" => Some(ProgramClass::V3), + "v4" => Some(ProgramClass::V4), + "v5" => Some(ProgramClass::V5), + _ => None, + } + } + + /// Whether the class reads the era seed (every class after v2: the era layout is drawn inside the class). + pub fn has_era(&self) -> bool { + *self != ProgramClass::V2 + } + + /// The program class whose load class `class` is, the era draw set aside: [`LoadClass::V2`] is v2, [`V3_CLASS`] + /// is v3, [`V4_CLASS`] is v4; a measurement class (a width, a derivation length, another shadow size) is none. + pub fn of_load_class(class: &LoadClass) -> Option { + let base = LoadClass { era: None, ..*class }; + if base == LoadClass::V2 { + Some(ProgramClass::V2) + } else if base == V3_CLASS { + Some(ProgramClass::V3) + } else if base == V4_CLASS { + Some(ProgramClass::V4) + } else if base == V5_CLASS { + Some(ProgramClass::V5) + } else { + None + } + } + + /// The generator version as a byte, for wire formats that carry the class as a number. + pub fn as_u8(&self) -> u8 { + self.generator_version() as u8 + } + + pub fn from_u8(v: u8) -> Option { + Self::from_generator(v as u32) + } +} + +impl Program { + pub fn loads_per_hash(&self) -> usize { + self.instrs.iter().filter(|i| i.op.is_load()).count() * ITERATIONS + } + pub fn wide_loads_per_hash(&self) -> usize { + self.instrs.iter().filter(|i| i.op == Op::WLoad).count() * ITERATIONS + } + pub fn has_wide(&self) -> bool { + self.instrs.iter().any(|i| i.op == Op::WLoad) + } + /// Dataset bytes read per hash: 4 per one-word load, 16 and 64 for the wider loads of the experiment. + pub fn bytes_per_hash(&self) -> usize { + self.instrs.iter().filter(|i| i.op == Op::Load).map(|i| i.width as usize * 4).sum::() * ITERATIONS + } + /// Scratch read-modify-writes per hash (variant 5): each reads 16 bytes and writes 16 bytes. + pub fn scratch_ops_per_hash(&self) -> usize { + self.instrs.iter().filter(|i| i.op == Op::Scratch).count() * ITERATIONS + } + pub fn has_scratch(&self) -> bool { + self.class.scratch.is_some() + } + /// Hot-table loads per hash (one 4-byte word each). + pub fn hot_loads_per_hash(&self) -> usize { + self.instrs.iter().filter(|i| i.op == Op::Hot).count() * ITERATIONS + } + pub fn has_hot(&self) -> bool { + self.class.hot.is_some() + } + /// Whether the program carries a latency-shadow block (Counter ASIC 3.0 item 8). + pub fn has_shadow(&self) -> bool { + self.class.shadow.is_some() && !self.shadow.is_empty() + } + /// Times the shadow block runs per iteration (0 without one). + pub fn shadow_reps(&self) -> usize { + self.class.shadow.map(|s| s.reps as usize).unwrap_or(0) + } + /// Shadow instructions executed per hash: `ITERATIONS x block x reps` (0 without a shadow). + pub fn shadow_instrs_per_hash(&self) -> usize { + self.shadow.len() * self.shadow_reps() * ITERATIONS + } + /// Op histogram of the shadow block, count descending then name ascending (empty without a shadow). + pub fn shadow_histogram(&self) -> Vec<(&'static str, usize)> { + let mut counts: Vec<(&'static str, usize)> = Vec::new(); + for i in &self.shadow { + let name = i.op.name(); + match counts.iter_mut().find(|(n, _)| *n == name) { + Some(e) => e.1 += 1, + None => counts.push((name, 1)), + } + } + counts.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(b.0))); + counts + } + /// The hot table's words (0 without one). + pub fn hot_words(&self) -> u32 { + self.class.hot.map(|h| crate::memhard::hot_words(h.mb as u32)).unwrap_or(0) + } + /// Width histogram of the loads, in words: (1, 4, 16) counts. + pub fn width_counts(&self) -> [usize; 3] { + let mut c = [0usize; 3]; + for i in self.instrs.iter().filter(|i| i.op == Op::Load) { + if let Some(k) = WIDTH_WORDS.iter().position(|&w| w == i.width) { + c[k] += 1; + } + } + c + } + /// Distinct dataset items a 32-lane warp touches per hash: 32 per plain load, 2 per wide load; scratch and + /// hot loads touch none (so the added form keeps 4,096). + pub fn items_per_warp(&self) -> usize { + (self.loads_per_hash() - self.wide_loads_per_hash() - self.scratch_ops_per_hash() - self.hot_loads_per_hash()) * 32 + + self.wide_loads_per_hash() * 2 + } + /// Op histogram, count descending then name ascending. + pub fn histogram(&self) -> Vec<(&'static str, usize)> { + let mut counts: Vec<(&'static str, usize)> = Vec::new(); + for i in &self.instrs { + let name = i.op.name(); + match counts.iter_mut().find(|(n, _)| *n == name) { + Some(e) => e.1 += 1, + None => counts.push((name, 1)), + } + } + counts.sort_by(|a, b| b.1.cmp(&a.1).then_with(|| a.0.cmp(b.0))); + counts + } + /// "load=16 add=8 ..." as written into program.h. + pub fn op_mix(&self) -> String { + self.histogram().iter().map(|(n, c)| format!("{n}={c}")).collect::>().join(" ") + } + /// The shadow block's op mix in the same form (empty without a shadow). + pub fn shadow_op_mix(&self) -> String { + self.shadow_histogram().iter().map(|(n, c)| format!("{n}={c}")).collect::>().join(" ") + } + /// The program id: FNV-1a 64 over `"igneum-program/" || generator_le32 || seed words as little-endian bytes + /// || attempt_le32`. Written into every pack so a version 1 program, or another attempt of the same seed, + /// can never be mistaken for this one. + pub fn program_id(&self) -> u64 { + // Latency ladder (docs/design/latency-ladder.md section 7): a class v4 program above rung 0 carries its shadow + // size in the id (`program_id_class`, the "shadow/" bytes), so two rungs of one seed never share an id and a + // pack of another rung is refused as a pack of another class is. Rung 0 keeps `program_id(4, seed, attempt)` + // byte for byte, so every v4 id written before the ladder stands. + let v4_rung_0 = self.generator == GENERATOR_VERSION_V4 && LoadClass { era: None, ..self.class } == V4_CLASS; + // class v5 at rung 0 is `program_id(5, seed, attempt)`; above rung 0 the class-bearing id with "state/" + let v5_rung_0 = self.generator == GENERATOR_VERSION_V5 && LoadClass { era: None, ..self.class } == V5_CLASS; + if self.class.is_v2() || self.generator == GENERATOR_VERSION_V3 || v4_rung_0 || v5_rung_0 { + // Spec 01 section 1.4.6: a class v3 program's id is `program_id(3, seed, attempt)`, a class v4 program's + // `program_id(4, seed, attempt)` (Counter ASIC 3.0); the generator version in the preimage separates + // them from every version 2 program of the same seed + program_id(self.generator, &self.seed, self.attempt) + } else { + program_id_class(self.generator, &self.seed, self.attempt, &self.class) + } + } + + /// The program class of this program, from its generator version (3 = v3, 4 = v4, everything else v2). + pub fn program_class(&self) -> ProgramClass { + match self.generator { + GENERATOR_VERSION_V3 => ProgramClass::V3, + GENERATOR_VERSION_V4 => ProgramClass::V4, + GENERATOR_VERSION_V5 => ProgramClass::V5, + _ => ProgramClass::V2, + } + } +} + +pub fn program_id(generator: u32, seed: &[u32; 8], attempt: u32) -> u64 { + let mut b = Vec::with_capacity(PROGRAM_ID_TAG.len() + 4 + 32 + 4 + 6); + b.extend_from_slice(PROGRAM_ID_TAG); + b.extend_from_slice(&generator.to_le_bytes()); + for w in seed { + b.extend_from_slice(&w.to_le_bytes()); + } + b.extend_from_slice(&attempt.to_le_bytes()); + if generator == GENERATOR_VERSION_V4 { + // The class v4 sub-version (AP-F8-1 amendment, 7 October 2026): `"sub/" || sub_version as little-endian u16` + // appended for generator 4 only, so a binary from before the load-source rule (sub-version 0, no suffix) and + // one after it never share a program id for one seed; the node's id check then catches a split. v2 and v3 + // ids are byte-identical. The node reads the sub-version from [`PROGRAM_SUBVERSION_V4`]; packs carry it as + // IGNEUM_PROGRAM_SUBVERSION and program.json "sub_version". + b.extend_from_slice(b"sub/"); + b.extend_from_slice(&PROGRAM_SUBVERSION_V4.to_le_bytes()); + } + fnv1a64(&b) +} + +/// The sub-version of class v4's program stream, in every generator-4 program id and pack (AP-F8-3: 3 = the +/// acceptance executing the shadow block as the hash does, so its verdicts judge the program the chain hashes; +/// 2 = the dataflow load-source rule and the saturated-source check (c'), never shipped; 1 = the one-writer rule, +/// the stream 0.3.20 and 0.3.21 ship as object byte 5; 0 was the stream of 6 October 2026, never stamped). +pub const PROGRAM_SUBVERSION_V4: u16 = 3; + +/// Domain tag of the program id of a read-width class (never collides with [`PROGRAM_ID_TAG`]). +pub const PROGRAM_ID_TAG_RW: &[u8] = b"igneum-program-rw/"; + +/// The program id of a non-default class: the tag, then the same fields as [`program_id`], then the three mix +/// percentages and the slot count as bytes. +pub fn program_id_class(generator: u32, seed: &[u32; 8], attempt: u32, class: &LoadClass) -> u64 { + let mut b = Vec::with_capacity(PROGRAM_ID_TAG_RW.len() + 4 + 32 + 4 + 4); + b.extend_from_slice(PROGRAM_ID_TAG_RW); + b.extend_from_slice(&generator.to_le_bytes()); + for w in seed { + b.extend_from_slice(&w.to_le_bytes()); + } + b.extend_from_slice(&attempt.to_le_bytes()); + b.extend_from_slice(&class.mix); + b.push(class.load_slots); + if let Some(k) = class.scratch { + b.extend_from_slice(b"scratch/"); + b.push(k); + b.push(class.scratch_kb); + } + if class.mixer_mult != 1 || class.growth { + // Counter ASIC 2.0: the mixer multiplier and the growth rule are part of the construction, so a program of + // the same seed under a different mixer carries a different id (under the v3 seam the id is + // program_id(3, seed, attempt) and this branch is not taken) + b.extend_from_slice(b"mixer/"); + b.push(class.mixer_mult); + b.push(class.growth as u8); + } + if class.derive_len != 0 { + // Counter ASIC 3.0 item 2: the derivation program's length is part of the construction + b.extend_from_slice(b"derive/"); + b.extend_from_slice(&class.derive_len.to_le_bytes()); + } + if let Some(e) = class.era { + b.extend_from_slice(b"era/"); + b.extend_from_slice(&e.id_bytes()); + } + if let Some(sh) = class.shadow { + // Counter ASIC 3.0 item 8: the shadow block's size and repeat count are part of the construction + b.extend_from_slice(b"shadow/"); + b.extend_from_slice(&sh.instrs.to_le_bytes()); + b.extend_from_slice(&sh.reps.to_le_bytes()); + } + if let Some(h) = class.hot { + b.extend_from_slice(b"hot/"); + b.push(h.mb); + b.push(h.k); + if h.added { + b.extend_from_slice(b"added"); + } + } + if class.state { + // class v5: the state leaves are part of the construction + b.extend_from_slice(b"state/"); + } + fnv1a64(&b) +} + +/// Weights of the ten non-load families under version 2, in draw order. Sum 75. The load family has no +/// weight: its count is fixed by [`LOAD_SLOTS`]. +pub const NONLOAD_WEIGHTS: [(Op, u64); 10] = [ + (Op::Add, 12), + (Op::Xor, 10), + (Op::Mul, 8), + (Op::Mad, 8), + (Op::Shfl, 8), + (Op::Rotl, 7), + (Op::Sub, 6), + (Op::MulHi, 6), + (Op::Rotr, 6), + (Op::Or, 4), +]; + +/// Version 1 weights (retired). Sum 100, load at 25 percent. +pub const OP_WEIGHTS: [(Op, u64); 11] = [ + (Op::Load, 25), + (Op::Add, 12), + (Op::Xor, 10), + (Op::Mul, 8), + (Op::Mad, 8), + (Op::Shfl, 8), + (Op::Rotl, 7), + (Op::Sub, 6), + (Op::MulHi, 6), + (Op::Rotr, 6), + (Op::Or, 4), +]; + +/// The seed words of attempt `k` of a program seed: `seed_words_from_bytes(seed_bytes)` for `k = 0`, +/// `seed_words_from_bytes(seed_bytes || k_le32)` otherwise. +pub fn attempt_words(seed_bytes: &[u8], attempt: u32) -> [u32; 8] { + if attempt == 0 { + return seed_words_from_bytes(seed_bytes); + } + let mut b = Vec::with_capacity(seed_bytes.len() + 4); + b.extend_from_slice(seed_bytes); + b.extend_from_slice(&attempt.to_le_bytes()); + seed_words_from_bytes(&b) +} + +/// One version 2 candidate from its seed words, before the acceptance rule. Spec 01 section 1.4.3: 16 slot draws, +/// then nine draws per instruction, 592 per program. +pub fn candidate_from_words(seed_string: &str, seed_bytes: &[u8], seed: [u32; 8], attempt: u32) -> Program { + candidate_from_words_class(seed_string, seed_bytes, seed, attempt, LoadClass::V2) +} + +/// [`candidate_from_words`] for a load class. For [`LoadClass::V2`] this is the version 2 draw stream exactly; +/// for any other class the slot count is the class's and every instruction takes a tenth draw, `below(100)`, +/// the width roll (used only on a load slot, drawn on every slot so the stream stays uniform). +/// Class v5's shadow rule (AP-F1-1): the peephole-removable share a block may have, per mille of its instructions. +pub const SHADOW_REMOVABLE_MAX_PERMILLE: usize = 30; +/// Class v5's shadow rule: redraws before the last block stands as drawn (never reached at 4e-3 per try). +pub const SHADOW_REDRAW_CAP: u32 = 64; + +/// The instructions of a shadow block an honest compiler removes (the AP-F1-1 census's classes): an instruction whose +/// destination's last writer, with no write to the destination or to the source in between, is the same op on the +/// same source and immediates and the pair cancels (xor: `x ^= s` twice) or is idempotent (or: `x |= s` twice), or a +/// rotate of a register last written by a rotate (the two merge into one), or an add followed by a sub (or a sub by an +/// add) of the same source and immediate (sum-cancel). Counted per removable instruction, never across a pass. +pub fn shadow_removable_count(block: &[Instr]) -> usize { + let mut last: [Option; 8] = [None; 8]; + let mut removable = 0; + for (k, ins) in block.iter().enumerate() { + let d = ins.dst as usize; + if let Some(j) = last[d] { + let prev = &block[j]; + // the source must not have been written between j and k (its value is the same) + let src_untouched = !block[j + 1..k].iter().any(|i| i.dst == ins.src); + let same_operands = prev.src == ins.src && prev.imm == ins.imm && prev.imm2 == ins.imm2 && prev.src2 == ins.src2 && prev.rot == ins.rot && prev.bit == ins.bit && prev.mask == ins.mask; + let hit = match (prev.op, ins.op) { + (Op::Xor, Op::Xor) | (Op::Or, Op::Or) => same_operands && src_untouched, + // every op reads its own destination (`d = d op a`), so two rotates on one register always fold on + // paper; the census counts "written twice from one source": the same rotate with the same amount + // (rotl, an immediate) or the same amount register unwritten between (rotr), which keeps the + // metric at the census's 0.62 percent average instead of every rotate pair + (Op::Rotl, Op::Rotl) => prev.rot == ins.rot, + (Op::Rotr, Op::Rotr) => prev.src == ins.src && src_untouched, + (Op::Add, Op::Sub) | (Op::Sub, Op::Add) => same_operands && src_untouched, + _ => false, + }; + if hit { + removable += 1; + } + } + last[d] = Some(k); + } + removable +} + +pub fn candidate_from_words_class( + seed_string: &str, + seed_bytes: &[u8], + seed: [u32; 8], + attempt: u32, + class: LoadClass, +) -> Program { + let mut rng = program_rng(&seed); + let slots = class.load_slots as usize; + // (1) Load slots: a uniform subset of 1..63 by partial Fisher-Yates. Instruction 0 is never a load. + let mut p: [u8; INSTR_COUNT - 1] = [0; INSTR_COUNT - 1]; + for (i, slot) in p.iter_mut().enumerate() { + *slot = (i + 1) as u8; + } + for i in 0..slots { + let j = i + rng.below((INSTR_COUNT - 1 - i) as u64) as usize; + p.swap(i, j); + } + let mut is_load = [false; INSTR_COUNT]; + for &slot in &p[..slots] { + is_load[slot as usize] = true; + } + // Variant 5: the first k drawn load slots (a uniform k-subset, the draw order is random) are scratch ops. + let mut is_scratch = [false; INSTR_COUNT]; + for &slot in &p[..class.scratch_slots()] { + is_scratch[slot as usize] = true; + } + // Hot table: the next k drawn load slots after the scratch slots are hot loads (again a uniform subset). + let mut is_hot = [false; INSTR_COUNT]; + for &slot in &p[class.scratch_slots()..class.scratch_slots() + class.hot_slots()] { + is_hot[slot as usize] = true; + } + // (2) The instructions. `fresh[r]`: r was written by an earlier instruction and no load has read it since. + // Class v4's load-source rule (AP-F8-1, 7 October 2026, `docs/analysis/ca3-v4-uniform.md`; sub-version 2): + // a load's source is drawn only from registers that are FRESH by dataflow. Fresh at the program start (the init + // words are a per-lane hash of the nonce); after an op, the destination is fresh when: a load's source was fresh + // (a saturated source reads one fixed word and leaves a constant); add, sub, xor, mad or shfl had a fresh operand + // (dst or src); rotl or rotr rotated a fresh value (a rotate maps all-ones to all-ones); never after or, mul or + // mulhi (an or-written value is all-ones with probability (3/4)^32 per read, the 153x item of the finding; mul + // zeroes low bits; mulhi is dense near zero). Sub-version 1 looked one writer back and counted every load and + // rotate as fresh, which let an or-saturated value through a rotate or a load-after-load chain (F8's p6, p31). + // Keyed on the class v4 shape (the 256-instruction shadow block over the class v3 base, the pass count and the + // era set aside) on EVERY draw path, era or not, so a census through candidate_class reads the same stream as + // the chain; v2, v3 and every other class take no part. The draw order and the stream are otherwise the same. + // class v5 (docs/design/class-v5-stored-state.md) draws under the same rule: its state flag is set aside here too + let source_rule_v4 = matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. })) + && LoadClass { era: None, shadow: None, state: false, ..class } == LoadClass { shadow: None, ..V4_CLASS }; + let mut fresh = [false; 8]; + let mut fresh_value = [true; 8]; + // the shared-operand idiom (AP-F8-1, sub-version 3): after `or d |= s`, a later `xor d ^= s` or `sub d -= s` with + // s unwritten since is `d & ~s`; after `xor d ^= s`, a later `or d |= s` is `d | s`: lossy either way, though + // the second op would count as injecting on its own. `pair_op[d]` holds the (op, s) of the last or/xor on d + // while neither d nor s has been written since. + let mut pair_op: [Option<(Op, usize)>; 8] = [None; 8]; + let mut instrs = Vec::with_capacity(INSTR_COUNT); + for k in 0..INSTR_COUNT { + let mut roll = rng.below(75); + let mut op = Op::Add; + for &(o, w) in &NONLOAD_WEIGHTS { + if roll < w { + op = o; + break; + } + roll -= w; + } + if is_load[k] { + op = if is_scratch[k] { + Op::Scratch + } else if is_hot[k] { + Op::Hot + } else { + Op::Load + }; + } + let dst = rng.below(8); + let src = if op.is_load() { + let mut eligible = [0u64; 8]; + let mut n = 0usize; + for r in 0..8u64 { + if r != dst && fresh[r as usize] && (!source_rule_v4 || fresh_value[r as usize]) { + eligible[n] = r; + n += 1; + } + } + if n == 0 { + let a = rng.below(7); + if a >= dst { + a + 1 + } else { + a + } + } else { + eligible[rng.below(n as u64) as usize] + } + } else { + let a = rng.below(7); + if a >= dst { + a + 1 + } else { + a + } + }; + let b = rng.below(8); + let imm = rng.next() as u32; + let imm2 = rng.next() as u32; + let rot = 1 + rng.below(31) as u32; + let bit = rng.below(32); + let mask = 1u8 << rng.below(5); + // Version 2 loads take no width roll, so a mixer class with version 2 loads draws the version 2 program + let width = if class.takes_width_roll() { class.width_for_roll(rng.below(100)) } else { 1 }; + let width = if op == Op::Load { width } else { 1 }; + // Era layout, layer 8: two window draws per instruction (drawn on every slot, used on a load slot). + let (win, off) = if class.era.is_some() { + let k = rng.below(3) as u8; + let o = (rng.next() as u32 & ((1u32 << k) - 1)) as u8; + if op == Op::Load { + (k, o) + } else { + (0, 0) + } + } else { + (0, 0) + }; + if op.is_load() { + fresh[src as usize] = false; + } + fresh[dst as usize] = true; + let (d, a) = (dst as usize, src as usize); + let masked = matches!((pair_op[d], op), (Some((Op::Or, s)), Op::Xor) | (Some((Op::Or, s)), Op::Sub) | (Some((Op::Xor, s)), Op::Or) if s == a); + fresh_value[d] = !masked + && match op { + Op::Load | Op::WLoad | Op::Scratch | Op::Hot => fresh_value[a], + Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl => fresh_value[d] || fresh_value[a], + Op::Rotl | Op::Rotr => fresh_value[d], + Op::Or | Op::Mul | Op::MulHi => false, + }; + // a write to d sets or clears d's pair; a write to any register clears every pair that names it as operand + pair_op[d] = if matches!(op, Op::Or | Op::Xor) && !masked { Some((op, a)) } else { None }; + for r in 0..8 { + if r != d { + if let Some((_, s)) = pair_op[r] { + if s == d { + pair_op[r] = None; + } + } + } + } + instrs.push(Instr { op, dst: dst as u8, src: src as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask, width, win, off }); + } + // (3) The latency-shadow block (Counter ASIC 3.0 item 8): drawn after the base program from the same stream, so + // the 64 instructions above are the class's without the shadow, draw for draw. Every slot is an ALU slot: the + // op from the non-load table, the source as on an ALU slot, the same per-instruction draws (the width roll and + // the era windows included when the class takes them, drawn and ignored) so the stream shape is the program's. + let mut shadow = Vec::new(); + let mut shadow_redraws = 0u32; + if let Some(sh) = class.shadow { + // Class v5 (docs/design/class-v5-stored-state.md section 11, AP-F1-1): a shadow block whose peephole-removable + // share exceeds SHADOW_REMOVABLE_MAX_PERMILLE (3.0 percent of the block; the census's histogram puts 384 of 100,000 + // draws there) is redrawn from the continuing stream, so an honest compiler's simplification cannot take more + // than 3 percent of the shadow's useful work. Every other class keeps its first draw. + loop { + shadow.clear(); + for _ in 0..sh.instrs { + let mut roll = rng.below(75); + let mut op = Op::Add; + for &(o, w) in &NONLOAD_WEIGHTS { + if roll < w { + op = o; + break; + } + roll -= w; + } + let dst = rng.below(8); + let a = rng.below(7); + let src = if a >= dst { a + 1 } else { a }; + let b = rng.below(8); + let imm = rng.next() as u32; + let imm2 = rng.next() as u32; + let rot = 1 + rng.below(31) as u32; + let bit = rng.below(32); + let mask = 1u8 << rng.below(5); + if class.takes_width_roll() { + let _ = rng.below(100); + } + if class.era.is_some() { + let _ = rng.below(3); + let _ = rng.next(); + } + shadow.push(Instr { op, dst: dst as u8, src: src as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask, width: 1, win: 0, off: 0 }); + } + if !class.state || shadow_removable_count(&shadow) * 1000 <= sh.instrs as usize * SHADOW_REMOVABLE_MAX_PERMILLE || shadow_redraws >= SHADOW_REDRAW_CAP { + break; + } + shadow_redraws += 1; + } + } + let _ = shadow_redraws; + Program { + seed_string: seed_string.to_string(), + seed_bytes: seed_bytes.to_vec(), + seed, + generator: GENERATOR_VERSION, + attempt, + class, + era_bytes: None, + instrs, + shadow, + } +} + +/// Candidate `attempt` of a program seed, before the acceptance rule. +pub fn candidate(seed_string: &str, seed_bytes: &[u8], attempt: u32) -> Program { + candidate_from_words(seed_string, seed_bytes, attempt_words(seed_bytes, attempt), attempt) +} + +/// [`candidate`] for a load class. +pub fn candidate_class(seed_string: &str, seed_bytes: &[u8], attempt: u32, class: LoadClass) -> Program { + candidate_from_words_class(seed_string, seed_bytes, attempt_words(seed_bytes, attempt), attempt, class) +} + +/// Why no program could be derived from a seed. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct Exhausted { + pub seed_string: String, + pub attempts: u32, + pub last: Reject, +} + +impl std::fmt::Display for Exhausted { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "seed {:?}: {} consecutive candidates rejected, last: {}", self.seed_string, self.attempts, self.last) + } +} + +impl std::error::Error for Exhausted {} + +/// The program of a seed: the first accepted candidate over attempts `0, 1, 2, ...`, at most [`MAX_ATTEMPTS`]. +/// This is what the chain calls (`Epoch::from_seed_bytes`) with the 32-byte epoch seed, and what the packs call +/// with the UTF-8 of a seed string. +pub fn try_generate_from_seed_bytes(seed_string: &str, seed_bytes: &[u8]) -> Result { + try_generate_class(seed_string, seed_bytes, LoadClass::V2) +} + +/// [`try_generate_from_seed_bytes`] for a load class. +pub fn try_generate_class(seed_string: &str, seed_bytes: &[u8], class: LoadClass) -> Result { + let mut last = None; + let cap = max_attempts_for(&class); + for attempt in 0..cap { + let p = candidate_class(seed_string, seed_bytes, attempt, class); + match check(&p) { + Ok(_) => return Ok(p), + Err(r) => last = Some(r), + } + } + if crate::accept::is_class_v4_shape(&class) { + // AP-F8-2 (7 October 2026, main's ruling: the draw is total and no consensus path panics): a class v4 seed that + // exhausts its attempts takes the last-resort program, deterministic and accepted as drawn + return Ok(last_resort_v4(candidate_class(seed_string, seed_bytes, cap, class))); + } + Err(Exhausted { seed_string: seed_string.to_string(), attempts: cap, last: last.unwrap() }) +} + +/// The last-resort program of a class v4 seed whose [`MAX_ATTEMPTS_V4`] candidates were all rejected (AP-F8-2): +/// the candidate at attempt [`MAX_ATTEMPTS_V4`] with every `or`, `mul` and `mulhi` of its base program and its shadow +/// block rewritten to `xor` (dst, src and the other fields kept). With no lossy op left every register stays fresh by dataflow from the +/// init words on, so rule (a') holds by construction; the program is the seed's consensus program as drawn, with no +/// further check, so the draw is total. It is reached with probability about (2/3)^256 per epoch seed (the measured +/// (a') plus (c') rejection rate of about two thirds per attempt), under 1e-45: the chain never sees it, and a test +/// walks it on real rejected candidates so the path is known to run. +pub fn last_resort_v4(mut p: Program) -> Program { + // the shadow block runs at the end of every iteration and its own lossy ops feed the next iteration's loads + // (rule (a') walks base then shadow to its fixpoint), so both are rewritten + for i in p.instrs.iter_mut().chain(p.shadow.iter_mut()) { + if matches!(i.op, Op::Or | Op::Mul | Op::MulHi) { + i.op = Op::Xor; + } + } + p +} + +/// [`try_generate_from_seed_bytes`], treating exhaustion as the consensus fault it is. +pub fn generate_from_seed_bytes(seed_string: &str, seed_bytes: &[u8]) -> Program { + try_generate_from_seed_bytes(seed_string, seed_bytes).unwrap_or_else(|e| panic!("{e}")) +} + +/// [`generate_from_seed_bytes`] for a load class. +pub fn generate_from_seed_bytes_class(seed_string: &str, seed_bytes: &[u8], class: LoadClass) -> Program { + try_generate_class(seed_string, seed_bytes, class).unwrap_or_else(|e| panic!("{e}")) +} + +/// The program of a program class (what the chain calls through `Epoch::from_chain_seeds`): class v2 is +/// [`generate_from_seed_bytes`] exactly; class v3 draws from [`V3_CLASS`] and stamps generator version 3 on the +/// program, so its packs and its id say generator 3 (spec 01 sections 1.4.5 and 1.4.6). +pub fn generate_from_seed_bytes_program_class(seed_string: &str, seed_bytes: &[u8], class: ProgramClass, era_bytes: Option<&[u8]>) -> Program { + // Class v3 with an era seed: the era layout (docs/plans/era-layout.md) drawn from the era bytes inside V3_CLASS. + // Class v4 (Counter ASIC 3.0): the same draw inside V4_CLASS (V3_CLASS plus the shadow block), generator 4. + // Without era bytes (a template before the era is known) the bare class stands. + match (class, era_bytes) { + (ProgramClass::V3, Some(era)) => return generate_era(seed_string, seed_bytes, V3_CLASS, era, &V3_ALLOWED), + (ProgramClass::V4, Some(era)) => return generate_era_generator(seed_string, seed_bytes, V4_CLASS, era, &V3_ALLOWED, GENERATOR_VERSION_V4), + // Class v5: the same draw inside V5_CLASS (V4_CLASS plus the state flag, which the draw does not read), generator 5. + (ProgramClass::V5, Some(era)) => return generate_era_generator(seed_string, seed_bytes, V5_CLASS, era, &V3_ALLOWED, GENERATOR_VERSION_V5), + _ => {} + } + let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, class.load_class()); + p.generator = class.generator_version(); + // The era is a property of class v3 and v4 programs; a version 2 program never records one, so the pinned v2 + // packs and every v2 export stay byte-identical whatever the chain reports for the era + p.era_bytes = if class.has_era() { era_bytes.map(|b| b.to_vec()) } else { None }; + p +} + +/// [`generate_from_seed_bytes_program_class`] at a rung of the latency ladder (`docs/design/latency-ladder.md` +/// section 2): `shadow_reps` is the shadow pass count the chain's step gives the epoch, 0 for the class's own. Class +/// v4 at a rung above 0 draws from [`v4_class_at`] with the era inside and generator 4 stamped; every other class, +/// and class v4 at rung 0, is [`generate_from_seed_bytes_program_class`] byte for byte. The base program, the 16 loads +/// and the era draw do not move with the rung: only the pass count of the shadow block does. +pub fn generate_from_seed_bytes_program_class_shadow(seed_string: &str, seed_bytes: &[u8], class: ProgramClass, era_bytes: Option<&[u8]>, shadow_reps: u16) -> Program { + if !matches!(class, ProgramClass::V4 | ProgramClass::V5) || shadow_reps == 0 || shadow_reps == V4_SHADOW_REPS { + return generate_from_seed_bytes_program_class(seed_string, seed_bytes, class, era_bytes); + } + // class v5 at a rung: class v4's rung with the state flag, generator 5 + let (base, generator) = if class == ProgramClass::V5 { (v5_class_at(shadow_reps), GENERATOR_VERSION_V5) } else { (v4_class_at(shadow_reps), GENERATOR_VERSION_V4) }; + match era_bytes { + Some(era) => generate_era_generator(seed_string, seed_bytes, base, era, &V3_ALLOWED, generator), + None => { + let mut p = generate_from_seed_bytes_class(seed_string, seed_bytes, base); + p.generator = generator; + p.era_bytes = None; + p + } + } +} + +/// The program of a seed string (its UTF-8 bytes are the program seed). +pub fn generate(seed_string: &str) -> Program { + generate_from_seed_bytes(seed_string, seed_string.as_bytes()) +} + +/// [`generate`] for a load class. +pub fn generate_class(seed_string: &str, class: LoadClass) -> Program { + generate_from_seed_bytes_class(seed_string, seed_string.as_bytes(), class) +} + +/// Every candidate of a seed up to and including the accepted one, with each rejection. For reports and tests. +pub fn attempts(seed_string: &str, seed_bytes: &[u8]) -> Vec<(Program, Result<(), Reject>)> { + attempts_class(seed_string, seed_bytes, LoadClass::V2) +} + +/// [`attempts`] for a load class. +pub fn attempts_class(seed_string: &str, seed_bytes: &[u8], class: LoadClass) -> Vec<(Program, Result<(), Reject>)> { + let mut out = Vec::new(); + for attempt in 0..MAX_ATTEMPTS { + let p = candidate_class(seed_string, seed_bytes, attempt, class); + let verdict = check(&p).map(|_| ()); + let accepted = verdict.is_ok(); + out.push((p, verdict)); + if accepted { + break; + } + } + out +} + +// --------------------------------------------------------------------------------------------------------- +// Version 1 (retired 4 October 2026) +// --------------------------------------------------------------------------------------------------------- + +/// Version 1 levers (`proto-metal/MEMHARD.md` section 2.4). The defaults reproduce the version 1 generator exactly. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct GeneratorConfig { + /// Percent weight of the load op. + pub load_weight: u64, + /// Percent of load instructions emitted as warp-coalesced wide loads. + pub wide_frac: u64, +} + +impl Default for GeneratorConfig { + fn default() -> Self { + Self { load_weight: 25, wide_frac: 0 } + } +} + +impl GeneratorConfig { + /// Scaled weights: load gets `load_weight`, the other ten ops share the rest in their original + /// proportions, rounded by largest remainder so the table still sums to 100. + pub fn weights(&self) -> Vec<(Op, u64)> { + if self.load_weight == 25 { + return OP_WEIGHTS.to_vec(); + } + let others = &OP_WEIGHTS[1..]; + let total: u64 = others.iter().map(|w| w.1).sum(); // 75 + let budget = 100 - self.load_weight; + let mut scaled: Vec<(Op, u64, u64)> = + others.iter().map(|&(op, w)| (op, (w * budget) / total, (w * budget) % total)).collect(); + let mut sum: u64 = scaled.iter().map(|s| s.1).sum(); + let mut order: Vec = (0..scaled.len()).collect(); + order.sort_by(|&a, &b| scaled[b].2.cmp(&scaled[a].2).then_with(|| a.cmp(&b))); + let mut k = 0; + while sum < budget { + scaled[order[k]].1 += 1; + sum += 1; + k += 1; + } + let mut out = vec![(Op::Load, self.load_weight)]; + out.extend(scaled.iter().map(|s| (s.0, s.1))); + out + } +} + +/// The retired version 1 generator from seed words: op rolled per instruction against the 11-family table, +/// load count free, no acceptance rule. `generateProgramV1` in the Swift, draw for draw. +pub fn generate_v1_from_words(seed_string: &str, seed: [u32; 8], cfg: &GeneratorConfig) -> Program { + let mut rng = program_rng(&seed); + let weights = cfg.weights(); + let mut instrs = Vec::with_capacity(INSTR_COUNT); + for _ in 0..INSTR_COUNT { + let mut roll = rng.below(100); + let mut op = Op::Add; + for &(o, w) in &weights { + if roll < w { + op = o; + break; + } + roll -= w; + } + let dst = rng.below(8); + let mut a = rng.below(7); + if a >= dst { + a += 1; + } + let b = rng.below(8); + let imm = rng.next() as u32; + let imm2 = rng.next() as u32; + let rot = 1 + rng.below(31) as u32; + let bit = rng.below(32); + let mask = 1u8 << rng.below(5); + // Lever (b): the already-drawn selector bit decides whether a load is wide, so the stream is unchanged. + if op == Op::Load && bit * 100 < cfg.wide_frac * 32 { + op = Op::WLoad; + } + instrs.push(Instr { op, dst: dst as u8, src: a as u8, src2: b as u8, imm, imm2, rot, bit: bit as u8, mask, width: 1, win: 0, off: 0 }); + } + Program { + seed_string: seed_string.to_string(), + seed_bytes: seed_string.as_bytes().to_vec(), + seed, + generator: 1, + attempt: 0, + class: LoadClass::V2, + era_bytes: None, + instrs, + shadow: Vec::new(), + } +} + +/// The retired version 1 generator for a seed string. +pub fn generate_v1(seed_string: &str, cfg: &GeneratorConfig) -> Program { + generate_v1_from_words(seed_string, seed_words_from_bytes(seed_string.as_bytes()), cfg) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn default_weights_unchanged() { + assert_eq!(GeneratorConfig::default().weights(), OP_WEIGHTS.to_vec()); + assert_eq!(NONLOAD_WEIGHTS.iter().map(|w| w.1).sum::(), 75); + assert_eq!(&OP_WEIGHTS[1..], &NONLOAD_WEIGHTS[..]); + } + + #[test] + fn load_weight_17_table() { + // MEMHARD.md section 2.4: add=13 xor=11 mul=9 mad=9 shfl=9 rotl=8 sub=7 mulhi=7 rotr=6 or=4. + let w = GeneratorConfig { load_weight: 17, wide_frac: 0 }.weights(); + let expect = [ + (Op::Load, 17), + (Op::Add, 13), + (Op::Xor, 11), + (Op::Mul, 9), + (Op::Mad, 9), + (Op::Shfl, 9), + (Op::Rotl, 8), + (Op::Sub, 7), + (Op::MulHi, 7), + (Op::Rotr, 6), + (Op::Or, 4), + ]; + assert_eq!(w, expect.to_vec()); + assert_eq!(w.iter().map(|x| x.1).sum::(), 100); + } + + #[test] + fn v1_genesis_shape() { + let p = generate_v1("igneum-genesis", &GeneratorConfig::default()); + assert_eq!(p.generator, 1); + assert_eq!(p.loads_per_hash(), 104); + assert_eq!(p.op_mix(), "load=13 xor=13 sub=7 shfl=6 add=5 mulhi=5 mad=4 rotr=4 mul=3 rotl=3 or=1"); + } + + /// Every version 2 candidate has 16 loads, none at instruction 0, and honours the generator contract. + #[test] + fn v2_shape_and_contract() { + for i in 0..200u32 { + let s = format!("igneum-shape/{i}"); + let p = candidate(&s, s.as_bytes(), 0); + assert_eq!(p.generator, GENERATOR_VERSION); + assert_eq!(p.instrs.len(), INSTR_COUNT); + assert_eq!(p.loads_per_hash(), 8 * LOAD_SLOTS); + assert_ne!(p.instrs[0].op, Op::Load); + assert!(!p.has_wide()); + for ins in &p.instrs { + assert_ne!(ins.dst, ins.src); + assert!((1..=31).contains(&ins.rot)); + assert!(ins.mask.is_power_of_two() && ins.mask <= 16); + } + } + } + + /// The fresh-source rule by construction: unless E was empty, no load reads a register that an earlier load + /// read without a write in between, cyclically. + #[test] + fn v2_loads_are_fresh_unless_fallback() { + let mut fallbacks = 0; + for i in 0..500u32 { + let s = format!("igneum-fresh/{i}"); + let p = candidate(&s, s.as_bytes(), 0); + let stale = crate::accept::check_static(&p).err(); + if let Some(Reject::StaleLoadSource { .. }) = stale { + fallbacks += 1; + } + } + // The census measured about 1.6 percent of candidates with a static repeat. + assert!(fallbacks < 30, "{fallbacks} of 500 candidates with a stale load"); + } + + #[test] + fn attempt_words_differ_and_are_stable() { + let a0 = attempt_words(b"igneum-genesis", 0); + assert_eq!(a0, seed_words_from_bytes(b"igneum-genesis")); + let a1 = attempt_words(b"igneum-genesis", 1); + assert_ne!(a0, a1); + assert_eq!(a1, seed_words_from_bytes(b"igneum-genesis\x01\x00\x00\x00")); + } + + #[test] + fn program_id_separates_versions_and_attempts() { + let p = generate("igneum-genesis"); + let v1 = generate_v1("igneum-genesis", &GeneratorConfig::default()); + assert_ne!(p.program_id(), v1.program_id()); + assert_ne!(program_id(2, &p.seed, 0), program_id(2, &p.seed, 1)); + } + + /// The read-width classes (5 October 2026): the default class is the version 2 stream exactly; a class + /// program has its slot count, widths from its mix only, and an id that separates it from version 2 and from + /// the other classes. + #[test] + fn load_classes() { + let v2 = candidate("igneum-genesis", b"igneum-genesis", 0); + let same = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::V2); + assert_eq!(v2, same); + assert!(v2.instrs.iter().all(|i| i.width == 1)); + assert_eq!(v2.bytes_per_hash(), 512); + assert_eq!(LoadClass::parse("w16"), Some(LoadClass::fixed(4, 16))); + assert_eq!(LoadClass::parse("w64x4"), Some(LoadClass::fixed(16, 4))); + assert_eq!(LoadClass::parse("50,35,15"), Some(LoadClass::mixed([50, 35, 15]))); + assert_eq!(LoadClass::parse("v2"), Some(LoadClass::V2)); + assert_eq!(LoadClass::parse("50,35,10"), None); + assert_eq!(LoadClass::fixed(16, 4).name(), "w64x4"); + assert_eq!(LoadClass::mixed([25, 50, 25]).name(), "mix25-50-25"); + // W = 4 with 16 slots IS the lottery hash: the w4 name parses to the default class + assert_eq!(LoadClass::fixed(1, 16), LoadClass::V2); + assert_eq!(LoadClass::parse("w4"), Some(LoadClass::V2)); + assert_eq!(LoadClass::fixed(1, 16).name(), "v2"); + assert_eq!(LoadClass::fixed(1, 8).name(), "w4x8"); + assert!((LoadClass::mixed([50, 35, 15]).expected_bytes_per_hash() - 2201.6).abs() < 1e-6); + assert!((LoadClass::fixed(16, 4).expected_bytes_per_hash() - 2048.0).abs() < 1e-9); + let mut ids = std::collections::HashSet::new(); + ids.insert(v2.program_id()); + for (name, slots, widths) in [("w4x8", 8, vec![1u8]), ("w16", 16, vec![4]), ("w64", 16, vec![16]), ("w64x4", 4, vec![16]), ("50,35,15", 16, vec![1, 4, 16]), ("25,50,25", 16, vec![1, 4, 16])] { + let c = LoadClass::parse(name).unwrap(); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.class, c); + assert_eq!(p.instrs.len(), INSTR_COUNT); + assert_eq!(p.loads_per_hash(), 8 * slots); + assert_ne!(p.instrs[0].op, Op::Load); + for ins in &p.instrs { + if ins.op == Op::Load { + assert!(widths.contains(&ins.width), "{name}: width {}", ins.width); + } else { + assert_eq!(ins.width, 1); + } + } + assert!(ids.insert(p.program_id()), "{name}: program id collides"); + } + // variant 5: k scratch ops among the 16 memory operations, the rest one-word loads + for (k, kb) in [(0u8, 32u8), (2, 32), (4, 128), (8, 128)] { + let c = LoadClass::parse(&format!("scr{k}k{kb}")).unwrap(); + assert_eq!(c, LoadClass::scratch(k, kb)); + assert_eq!(c.name(), format!("scr{k}k{kb}")); + assert_eq!(c.scratch_bytes_per_warp(), kb as usize * 1024); + assert_eq!(c.scratch_slots_per_lane(), kb as usize * 2); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.loads_per_hash(), 128); + assert_eq!(p.scratch_ops_per_hash(), 8 * k as usize); + assert_eq!(p.bytes_per_hash(), (16 - k as usize) * 8 * 4); + assert!(p.instrs.iter().all(|i| i.width == 1)); + assert!(ids.insert(p.program_id()), "scr{k}: program id collides"); + } + assert!(!LoadClass::scratch(0, 32).is_v2()); + assert_ne!(LoadClass::scratch(4, 32).name(), LoadClass::scratch(4, 128).name()); + assert_eq!(LoadClass::parse("scr4"), None); + // a class with the version 2 widths but another slot count takes the extra roll: a different stream + let w4x8 = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::fixed(1, 8)); + assert_ne!(w4x8.instrs, v2.instrs); + // the mix draws every width over a population + let mut counts = [0usize; 3]; + for i in 0..200u32 { + let s = format!("igneum-rw-mix/{i}"); + let p = candidate_class(&s, s.as_bytes(), 0, LoadClass::mixed([50, 35, 15])); + let c = p.width_counts(); + for k in 0..3 { + counts[k] += c[k]; + } + } + let total = (counts[0] + counts[1] + counts[2]) as f64; + assert_eq!(total as usize, 200 * 16); + assert!((counts[0] as f64 / total - 0.50).abs() < 0.05, "{counts:?}"); + assert!((counts[1] as f64 / total - 0.35).abs() < 0.05, "{counts:?}"); + assert!((counts[2] as f64 / total - 0.15).abs() < 0.05, "{counts:?}"); + } + + /// Counter ASIC 2.0 seam: class v2 is the lottery hash exactly; class v3 is generator 3 on the placeholder load + /// class, with an id that no version 2 program of the seed can carry; the class round-trips through its name and + /// its generator number. + #[test] + fn program_classes() { + let v2 = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V2, Some(&[7u8; 32])); + let plain = generate("igneum-genesis"); + assert_eq!(v2, plain, "a v2 program never records an era"); + assert_eq!(v2.era_bytes, None); + assert_eq!(v2.program_class(), ProgramClass::V2); + assert_eq!(v2.generator, GENERATOR_VERSION); + let v3 = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, Some(&[7u8; 32])); + assert_eq!(v3.generator, GENERATOR_VERSION_V3); + assert_eq!(v3.era_bytes.as_deref(), Some(&[7u8; 32][..])); + let v3_no_era = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, None); + assert_ne!(v3_no_era.instrs, v3.instrs, "the era class takes two more draws per instruction"); + assert_eq!(v3_no_era.class, V3_CLASS); + assert_eq!(v3.program_class(), ProgramClass::V3); + assert_eq!(LoadClass { era: None, ..v3.class }, V3_CLASS, "the era rides inside V3_CLASS"); + assert_eq!(v3.class, LoadClass::era(V3_CLASS, &[7u8; 32], &V3_ALLOWED)); + assert_eq!(v3.class.era.unwrap().width_words, 1); + assert_eq!(v3.class.layout(), v3.class.era.unwrap().layout()); + assert_ne!(generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, Some(&[8u8; 32])).class, v3.class); + assert!(check(&v3).is_ok()); + assert_eq!(v3.program_id(), program_id(GENERATOR_VERSION_V3, &v3.seed, v3.attempt)); + assert_ne!(v3.program_id(), program_id(GENERATOR_VERSION, &v3.seed, v3.attempt)); + assert_ne!(v3.program_id(), v2.program_id()); + for c in [ProgramClass::V2, ProgramClass::V3, ProgramClass::V4, ProgramClass::V5] { + assert_eq!(ProgramClass::parse(c.name()), Some(c)); + assert_eq!(ProgramClass::from_generator(c.generator_version()), Some(c)); + assert_eq!(ProgramClass::from_u8(c.as_u8()), Some(c)); + } + assert_eq!(ProgramClass::from_generator(1), None); + assert_eq!(ProgramClass::from_generator(6), None); + assert_eq!(ProgramClass::parse("v6"), None); + assert_eq!(ProgramClass::default(), ProgramClass::V2); + assert_eq!(ProgramClass::V2.load_class(), LoadClass::V2); + assert!(!ProgramClass::V2.has_era() && ProgramClass::V3.has_era() && ProgramClass::V4.has_era() && ProgramClass::V5.has_era()); + assert!(!ProgramClass::V4.has_state() && ProgramClass::V5.has_state()); + } + + /// Class v5's shadow rule (AP-F1-1), the known-failed case first: a synthetic block of xor-cancel pairs is counted and + /// is over the bound; a scan of seeds finds a v5 draw whose first block was redrawn (the census's 4e-3), every v5 + /// block is under 3.0 percent removable, and the same seed's class v4 block (never redrawn) is the first draw. + #[test] + fn class_v5_shadow_redundancy_rule() { + let mk = |op: Op, dst: u8, src: u8| Instr { op, dst, src, src2: 0, imm: 7, imm2: 9, rot: 3, bit: 0, mask: 1, width: 1, win: 0, off: 0 }; + let pair = vec![mk(Op::Xor, 1, 2), mk(Op::Xor, 1, 2)]; + assert_eq!(shadow_removable_count(&pair), 1, "the known-failed case: an xor-cancel pair"); + let broken = vec![mk(Op::Xor, 1, 2), mk(Op::Add, 2, 3), mk(Op::Xor, 1, 2)]; + assert_eq!(shadow_removable_count(&broken), 0, "the source moved between the two"); + let rot = vec![mk(Op::Rotl, 4, 0), mk(Op::Rotl, 4, 0)]; + assert_eq!(shadow_removable_count(&rot), 1, "two rotl by one amount merge"); + let mixed = vec![mk(Op::Rotl, 4, 0), mk(Op::Rotr, 4, 5)]; + assert_eq!(shadow_removable_count(&mixed), 0, "a rotl and a rotr are two sources: not the census's merge"); + let rotr = vec![mk(Op::Rotr, 4, 5), mk(Op::Rotr, 4, 5)]; + assert_eq!(shadow_removable_count(&rotr), 1, "two rotr by one register merge"); + let sum = vec![mk(Op::Add, 6, 7), mk(Op::Sub, 6, 7)]; + assert_eq!(shadow_removable_count(&sum), 1, "sum-cancel"); + let mut bad = Vec::new(); + for _ in 0..128 { + bad.push(mk(Op::Or, 3, 5)); + bad.push(mk(Op::Or, 3, 5)); + } + assert!(shadow_removable_count(&bad) * 1000 > bad.len() * SHADOW_REMOVABLE_MAX_PERMILLE, "a block of or-idempotent pairs is over the bound"); + let era = [7u8; 32]; + let mut redrawn = 0; + let mut scanned = 0; + for i in 0..6_000u32 { + let seed = format!("igneum-shadow-scan/{i}"); + let v5 = generate_from_seed_bytes_program_class(&seed, seed.as_bytes(), ProgramClass::V5, Some(&era)); + assert!(shadow_removable_count(&v5.shadow) * 1000 <= v5.shadow.len() * SHADOW_REMOVABLE_MAX_PERMILLE, "{seed}: a v5 block over the bound"); + let v4 = generate_from_seed_bytes_program_class(&seed, seed.as_bytes(), ProgramClass::V4, Some(&era)); + if v5.shadow == v4.shadow { + assert_eq!(v5.instrs, v4.instrs, "{seed}: an unredrawn seed is the amended v4 draw for draw"); + } else { + // the acceptance rules read the shadow (the freshness fixpoint), so a redrawn block can move the accepted + // attempt and with it the base program; what must hold is that the first v4 block was over the bound + redrawn += 1; + let first = candidate_class(&seed, seed.as_bytes(), v4.attempt, v4.class); + assert!(shadow_removable_count(&first.shadow) * 1000 > first.shadow.len() * SHADOW_REMOVABLE_MAX_PERMILLE || v5.attempt != v4.attempt, "{seed}: v5 redrew a block the rule admits"); + } + scanned += 1; + if redrawn >= 2 && scanned >= 1_000 { + break; + } + } + assert!(redrawn >= 1, "no redraw in {scanned} seeds (the census says about 4e-3 per draw)"); + assert!(redrawn * 50 <= scanned, "{redrawn} redraws in {scanned} seeds: the metric is far above the census's 4e-3"); + } + + /// Class v5 (docs/design/class-v5-stored-state.md) takes the amended class v4 draw (AP-F8-1) as the chain draws it: + /// with an era present every load's source was last written by an injecting op or a rotate, the base program and + /// the shadow block equal the amended v4's of the same seed, and the same holds at a ladder rung; the state flag + /// changes the id and the dataset, never the draw. The known-failed case first: a v5 draw with a lossy-sourced + /// load would fail the same scan the amended v4 passes. + #[test] + fn class_v5_chain_draw_is_the_amended_v4_draw() { + let era = [7u8; 32]; + let scan = |p: &Program| { + let mut kept = [false; 8]; + let mut lossy = 0; + for i in &p.instrs { + if i.op.is_load() && !kept[i.src as usize] { + lossy += 1; + } + kept[i.dst as usize] = i.op.injects() || matches!(i.op, Op::Rotl | Op::Rotr); + } + lossy + }; + for seed in ["igneum-genesis", "igneum-epoch-7", "igneum-epoch-99"] { + let v4 = generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V4, Some(&era)); + let v5 = generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V5, Some(&era)); + if v5.shadow == v4.shadow { + assert_eq!(v5.instrs, v4.instrs, "{seed}: the base program is the amended v4's"); + assert_eq!((v5.seed, v5.attempt), (v4.seed, v4.attempt)); + } else { + // the AP-F1-1 shadow rule redrew this seed's block (and the acceptance, which reads the shadow, may have + // moved the attempt): the first v4 block must have been over the bound + let first = candidate_class(seed, seed.as_bytes(), v4.attempt, v4.class); + assert!(shadow_removable_count(&first.shadow) * 1000 > first.shadow.len() * SHADOW_REMOVABLE_MAX_PERMILLE || v5.attempt != v4.attempt, "{seed}: v5 differs from v4 without a redraw"); + } + // the draw equality above is the claim; sub-version 3's own freshness rule (dataflow, with the last resort after + // the 256 cap) is what the chain applies, so the sub-version 1 scan below is informational for v5 + let _ = scan(&v5); + assert_eq!(v5.generator, GENERATOR_VERSION_V5); + assert!(v5.class.state && !v4.class.state); + assert_ne!(v5.program_id(), v4.program_id()); + let r1 = generate_from_seed_bytes_program_class_shadow(seed, seed.as_bytes(), ProgramClass::V5, Some(&era), 35); + let r1v4 = generate_from_seed_bytes_program_class_shadow(seed, seed.as_bytes(), ProgramClass::V4, Some(&era), 35); + if r1.shadow == r1v4.shadow { + assert_eq!(r1.instrs, r1v4.instrs, "{seed}: rung 1 too"); + } + let _ = scan(&r1); + assert_eq!(r1.class, v5_class_at(35).with_era_of(&r1v4.class)); + } + // the known-failed case: the unamended v3 stream of the same seeds is lossy-sourced somewhere in three seeds and is + // not the v5 draw (a v5 that drew without the rule would equal it) + let seeds = ["igneum-genesis", "igneum-epoch-7", "igneum-epoch-99"]; + let lossy_v3: usize = seeds.iter().map(|seed| scan(&generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V3, Some(&era)))).sum(); + assert!(lossy_v3 > 0, "the known-failed case: the unamended stream carries lossy-sourced loads"); + assert!(seeds.iter().any(|seed| generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V3, Some(&era)).instrs != generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V5, Some(&era)).instrs), "v5 is not the unamended stream"); + } + + /// Counter ASIC 3.0 (6 October 2026): class v4 is class v3 with the latency-shadow block `sh256x27`, drawn after + /// the base program; since the AP-F8-1 amendment (7 October 2026) its base program draws a load's source only + /// from registers whose last writer keeps entropy, so it is its own stream over class v3's load slots and era + /// draw; its generator is 4, its id `program_id(4, seed, attempt)` with the sub-version suffix, its era recorded; + /// v2 and v3 are untouched. + /// Every load's source fresh by dataflow in the loop's steady state: the crate's own rule (a') of `accept.rs`. + fn every_load_source_fresh(p: &Program) -> bool { + crate::accept::check_fresh_sources_v4(p).is_ok() + } + + /// Class v4 sub-version 3, rule (c''): the distinct-index ratio refuses F8's low-entropy band on the chain's own + /// candidates (the first attempt of each strong seed past (a), (b), (c) and (c') fails the 2^20 pass), and the + /// shared-operand rule removes p23's value constant at the draw: the chain's attempt 1 (id d65122675f16a1c7) draws + /// site 7 from r5 and passes; the same program with that source put back to r6 (`or r6 |= r4; xor r6 ^= r4` + /// upstream) is refused by the static rule and, run anyway, by the ratio. + #[test] + fn class_v4_distinct_ratio_rejects_the_low_entropy_band() { + use crate::accept::{check_dynamic, check_static, Reject}; + use crate::seed::seed_words_from_bytes; + let w = |s: String| -> Vec { seed_words_from_bytes(s.as_bytes()).iter().flat_map(|x| x.to_le_bytes()).collect() }; + let f8 = |k: u32| (w(format!("igneum-attack-f8/program/{k}")), w(format!("igneum-attack-f8/era/{k}"))); + let candidate = |epoch: &[u8], era: &[u8], attempt: u32| { + let label = format!("igneum-epoch/{}", epoch.iter().map(|b| format!("{b:02x}")).collect::()); + let mut p = candidate_class(&label, epoch, attempt, LoadClass::era(V4_CLASS, era, &V3_ALLOWED)); + p.generator = GENERATOR_VERSION_V4; + p.era_bytes = Some(era.to_vec()); + p + }; + for k in [15u32, 18, 19, 56] { + let (epoch, era) = f8(k); + let mut seen = None; + for attempt in 0..MAX_ATTEMPTS_V4 { + let p = candidate(&epoch, &era, attempt); + let t0 = std::time::Instant::now(); + match check_static(&p).and_then(|_| check_dynamic(&p).map(|_| ())) { + Err(Reject::LowEntropySite { site, distinct, evaluations, ratio_milli }) => { + println!("p{k} attempt {attempt} id {:016x}: (c'') site {site} read {distinct} distinct over {evaluations}, ratio {}.{:03}, {:.1} s", p.program_id(), ratio_milli / 1000, ratio_milli % 1000, t0.elapsed().as_secs_f64()); + seen = Some(attempt); + break; + } + Err(_) => continue, + Ok(()) => break, + } + } + assert!(seen.is_some(), "p{k}: the chain reaches a candidate only the ratio refuses"); + } + let (epoch, era) = f8(23); + let p = candidate(&epoch, &era, 1); + assert_eq!(p.program_id(), 0xd65122675f16a1c7); + let site7 = p.instrs.iter().enumerate().filter(|(_, i)| i.op.is_load()).nth(7).map(|(k, _)| k).unwrap(); + println!("p23 attempt 1: site 7 is instruction {site7}, source r{}", p.instrs[site7].src); + assert_eq!(p.instrs[site7].src, 5, "the shared-operand rule moved site 7 off r6"); + assert!(check_static(&p).is_ok(), "p23 attempt 1 passes the static rule"); + let t0 = std::time::Instant::now(); + let v = check_dynamic(&p).map(|_| ()); + println!("p23 attempt 1 dynamic: {:?} in {:.1} s", v.as_ref().err().map(|x| x.to_string()), t0.elapsed().as_secs_f64()); + assert!(v.is_ok(), "p23 attempt 1 passes the dynamic rule"); + let mut q = p.clone(); + q.instrs[site7].src = 6; + assert!(matches!(check_static(&q), Err(Reject::UnfreshLoadSource { .. })), "the value constant's load is refused by the static rule"); + let v = check_dynamic(&q).map(|_| ()); + println!("p23 attempt 1 with site 7 from r6: {:?}", v.as_ref().err().map(|x| x.to_string())); + assert!(matches!(v, Err(Reject::LowEntropySite { .. })), "and by the ratio when run"); + } + + /// Diagnostic (AP-F8-1, p23): the distinct word indices per site on the closed-form words at 2^20 and 2^24 + /// evaluations, for the program F8 measured (attempt 1, id d64dbc675f13be9e): site 7 (instruction 38) reads + /// r6 = (mulhi(..) | r4) ^ r4 = r6 & ~r4, an andnot idiom the lineage rule counts as fresh. + #[test] + #[ignore] + fn diag_p23_distinct_indices_per_site() { + let hx = |h: &str| -> Vec { (0..h.len()).step_by(2).map(|i| u8::from_str_radix(&h[i..i + 2], 16).unwrap()).collect() }; + let e = hx("01aa1485fcb5d59223ca40602086e618277decf440188c5a55898f635f7be34f"); + let r = hx("00951c99e7ef952fd52611b8de7c07cc705cdec9e129a31a19d696c99909f052"); + let mut p = candidate_class("igneum-epoch/01aa1485fcb5d59223ca40602086e618277decf440188c5a55898f635f7be34f", &e, 1, LoadClass::era(V4_CLASS, &r, &V3_ALLOWED)); + p.generator = GENERATOR_VERSION_V4; + p.era_bytes = Some(r.clone()); + assert_eq!(p.program_id(), 0xd64dbc675f13be9e); + for units in [4096usize, 65536] { + let t0 = std::time::Instant::now(); + let d = crate::accept::distinct_indices_v4(&p, units).unwrap(); + println!("p23 d64dbc675f13be9e at {} evaluations per site ({:.1} s): distinct word indices per site {:?}; site 7 = {} (log2 {:.1})", units * 32 * 8, t0.elapsed().as_secs_f64(), d, d[7], (d[7] as f64).log2()); + } + } + + /// The threshold measurement for the second sub-version 3 commit: every F8 program (p1 = the devnet epoch-0 seeds, + /// p2 to p64 = the attack-pass harness's label-derived seeds), drawn under this commit's verdicts, each load + /// site's distinct word indices over 2^20 evaluations against the window expectation N - N^2 / 2W, the minimum + /// ratio per seed. Clean seeds set the threshold; the failing seeds of F8's table must sit below it. + #[test] + #[ignore] + fn diag_f8_64_distinct_index_ratios() { + use crate::seed::seed_words_from_bytes; + let w = |s: String| -> Vec { seed_words_from_bytes(s.as_bytes()).iter().flat_map(|x| x.to_le_bytes()).collect() }; + let hx = |h: &str| -> Vec { (0..h.len()).step_by(2).map(|i| u8::from_str_radix(&h[i..i + 2], 16).unwrap()).collect() }; + let n = 1u64 << 20; + for k in 1..=64u32 { + let (epoch, era) = if k == 1 { + let g = hx("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07"); + (g.clone(), g) + } else { + (w(format!("igneum-attack-f8/program/{k}")), w(format!("igneum-attack-f8/era/{k}"))) + }; + let label = format!("igneum-epoch/{}", epoch.iter().map(|b| format!("{b:02x}")).collect::()); + let p = generate_era(&label, &epoch, V4_CLASS, &era, &V3_ALLOWED); + let t0 = std::time::Instant::now(); + let d = crate::accept::distinct_indices_v4(&p, 4096).unwrap(); + let mut ratios = Vec::new(); + let mut site = 0usize; + for i in &p.instrs { + if !i.op.is_load() { continue; } + let k_off = (i.win as u64).min(2); + let wsize = (1u64 << 28) >> k_off; + let expected = n as f64 - (n as f64) * (n as f64) / (2.0 * wsize as f64); + ratios.push((d[site] as f64 / expected, site, i.win)); + site += 1; + } + let (min_ratio, min_site, min_win) = ratios.iter().cloned().fold((9.0, 0, 0), |a, b| if b.0 < a.0 { b } else { a }); + println!("RATIO p{k} attempt {} id {:016x}: min {:.4} at site {} (win {}) ; all {} ; {:.1} s", p.attempt, p.program_id(), min_ratio, min_site, min_win, ratios.iter().map(|r| format!("{:.3}", r.0)).collect::>().join(" "), t0.elapsed().as_secs_f64()); + } + } + + /// The 2^24 reach of the ratio for the weak failing seeds (p34, p4, p8, p10 at 1.22x to 1.50x sit inside the + /// clean spread at 2^20), with p23 and five clean seeds as the scale. + #[test] + #[ignore] + fn diag_f8_weak_seeds_at_2e24() { + use crate::seed::seed_words_from_bytes; + let w = |s: String| -> Vec { seed_words_from_bytes(s.as_bytes()).iter().flat_map(|x| x.to_le_bytes()).collect() }; + let n = 1u64 << 24; + for k in [34u32, 4, 8, 10, 23, 2, 3, 5, 44, 52] { + let (epoch, era) = (w(format!("igneum-attack-f8/program/{k}")), w(format!("igneum-attack-f8/era/{k}"))); + let label = format!("igneum-epoch/{}", epoch.iter().map(|b| format!("{b:02x}")).collect::()); + let p = generate_era(&label, &epoch, V4_CLASS, &era, &V3_ALLOWED); + let t0 = std::time::Instant::now(); + let d = crate::accept::distinct_indices_v4(&p, 65536).unwrap(); + let mut ratios = Vec::new(); + let mut site = 0usize; + for i in &p.instrs { + if !i.op.is_load() { continue; } + let wsize = (1u64 << 28) >> (i.win as u64).min(2); + let expected = n as f64 - (n as f64) * (n as f64) / (2.0 * wsize as f64); + ratios.push(d[site] as f64 / expected); + site += 1; + } + let min = ratios.iter().cloned().fold(9.0f64, f64::min); + println!("RATIO24 p{k} attempt {} id {:016x}: min {:.4} ; all {} ; {:.1} s", p.attempt, p.program_id(), min, ratios.iter().map(|r| format!("{:.3}", r)).collect::>().join(" "), t0.elapsed().as_secs_f64()); + } + } + + #[test] + fn class_v4_draw_is_total_with_the_last_resort() { + assert_eq!(max_attempts_for(&V4_CLASS), MAX_ATTEMPTS_V4); + assert_eq!(max_attempts_for(&LoadClass::era(V4_CLASS, &[7u8; 32], &V3_ALLOWED)), MAX_ATTEMPTS_V4); + assert_eq!(max_attempts_for(&V3_CLASS), MAX_ATTEMPTS); + assert_eq!(max_attempts_for(&LoadClass::V2), MAX_ATTEMPTS); + let class = LoadClass::era(V4_CLASS, &EraParams::test_era_bytes("igneum-era-test/0"), &V3_ALLOWED); + // real candidates that rule (a') rejects (the exhausting shape of seed igneum-f9/331672 at 07a809a7: 32 in a + // row), repaired by the last resort: every load's source fresh in the loop's steady state + let mut rejected = 0; + for i in 0..64u32 { + let seed = format!("igneum-ca3-v4-amend/total/{i}"); + for attempt in 0..4u32 { + let c = candidate_class(&seed, seed.as_bytes(), attempt, class); + if matches!(crate::accept::check_fresh_sources_v4(&c), Err(crate::accept::Reject::UnfreshLoadSource { .. })) { + rejected += 1; + let fixed = last_resort_v4(c.clone()); + assert!(crate::accept::check_fresh_sources_v4(&fixed).is_ok(), "{seed} attempt {attempt}: the last resort is fresh at every load"); + assert!(fixed.instrs.iter().chain(fixed.shadow.iter()).all(|i| !matches!(i.op, Op::Or | Op::Mul | Op::MulHi))); + assert_eq!((fixed.instrs.len(), fixed.shadow.len()), (c.instrs.len(), c.shadow.len())); + } + } + } + assert!(rejected > 0, "the sample holds (a')-rejected candidates (about two thirds of attempts do)"); + // the chain path is total: 64 seeds, every one a program + for i in 0..64u32 { + let seed = format!("igneum-ca3-v4-amend/total/{i}"); + let p = try_generate_class(&seed, seed.as_bytes(), class).expect("a class v4 seed always draws"); + assert!(crate::accept::check_fresh_sources_v4(&p).is_ok()); + assert!(p.attempt < MAX_ATTEMPTS_V4 || p.instrs.iter().chain(p.shadow.iter()).all(|i| !matches!(i.op, Op::Or | Op::Mul | Op::MulHi))); + } + } + + #[test] + fn program_class_v4_is_class_v3_with_the_shadow_block() { + assert_eq!(V4_CLASS, V3_CLASS.with_shadow(256, 27)); + assert_eq!(V4_CLASS.name(), "mx8+sh256x27"); + assert_eq!(LoadClass::parse("mx8+sh256x27"), Some(V4_CLASS)); + assert_eq!(LoadClass { shadow: None, ..V4_CLASS }, V3_CLASS, "v4 differs from v3 in the shadow alone"); + assert_eq!(V4_CLASS.shadow_instrs_per_hash(), 8 * 256 * 27); + assert_eq!(ProgramClass::V4.load_class(), V4_CLASS); + assert_eq!(ProgramClass::V4.generator_version(), GENERATOR_VERSION_V4); + let era = [7u8; 32]; + let v3 = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, Some(&era)); + let v4 = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era)); + assert_eq!(v4.generator, GENERATOR_VERSION_V4); + assert_eq!(v4.program_class(), ProgramClass::V4); + assert_eq!(v4.era_bytes.as_deref(), Some(&era[..])); + // The AP-F8-1 amendment (7 October 2026): class v4's chain draw takes a load's source only from registers + // whose last writer injects or rotates, so its base program is its own stream (the 6 October stream, equal + // to class v3's draw for draw, is sub-version 0 and never stamped); the load slots, the op draws and the era + // draw are still class v3's, and every load site obeys the rule + // sub-version 2 rejects candidates (rules (a') and (c')), so the accepted attempt can differ from class v3's and + // the seed words carry the attempt: compare with the class v3 candidate at the SAME attempt + let v3c = candidate_class("igneum-genesis", b"igneum-genesis", v4.attempt, LoadClass::era(V3_CLASS, &era, &V3_ALLOWED)); + assert_eq!(v4.seed, v3c.seed); + assert_eq!( + v4.instrs.iter().map(|i| i.op.is_load()).collect::>(), + v3c.instrs.iter().map(|i| i.op.is_load()).collect::>(), + "the load slots are class v3's at the same attempt" + ); + assert!(every_load_source_fresh(&v4), "every load of the amended v4 program reads a fresh register"); + let v3c_as_v4 = Program { class: v4.class, shadow: v4.shadow.clone(), ..v3c.clone() }; + if !every_load_source_fresh(&v3c_as_v4) { + assert_ne!(v4.instrs, v3c.instrs, "the amended v4 base program is not class v3's (a lossy-sourced load was redrawn)"); + } + // the same seed without an era draws under the rule too (sub-version 2: the rule is keyed on the class shape on + // every draw path), and the v3 program of the seed is the known-failed case + let v4_no_era = generate_from_seed_bytes_class("igneum-genesis", b"igneum-genesis", V4_CLASS); + assert!(every_load_source_fresh(&v4_no_era)); + // the known-failed case: the v3 program of the same seed and era, re-labelled with the v4 shape so the rule + // applies, carries an unfresh load source (96.6 percent of chain-shaped seeds do, ca3-v4-uniform.md section 3) + let v3_as_v4 = Program { class: v4.class, shadow: v4.shadow.clone(), ..v3.clone() }; + println!("class v3 of igneum-genesis under era [7; 32] (attempt {}): every load source fresh = {}; v4 accepted at attempt {}", v3.attempt, every_load_source_fresh(&v3_as_v4), v4.attempt); + assert!(v3.shadow.is_empty() && !v3.has_shadow()); + assert_eq!(v4.shadow.len(), 256); + assert_eq!(v4.shadow_reps(), 27); + assert_eq!(v4.shadow_instrs_per_hash(), 55_296); + assert_eq!(LoadClass { era: None, ..v4.class }, V4_CLASS, "the era rides inside V4_CLASS"); + assert_eq!(v4.class.era, v3.class.era, "the same era draw as class v3"); + assert_eq!(v4.class, LoadClass::era(V4_CLASS, &era, &V3_ALLOWED)); + assert!(check(&v4).is_ok()); + assert_eq!(v4.program_id(), program_id(GENERATOR_VERSION_V4, &v4.seed, v4.attempt)); + assert_ne!(v4.program_id(), v3.program_id()); + assert_ne!(v4.program_id(), program_id(GENERATOR_VERSION, &v4.seed, v4.attempt)); + // the CLI's `--era` path (generate_era over a base class) stamps 4 on V4_CLASS and 3 on anything else, so a + // pack exported with `--class mx8+sh256x27 --era` is a class v4 pack with the v4 id (the 6 October trap: + // the seven gate packs stamped 3 carried the v3 control's id) + let via_era = generate_era("igneum-genesis", b"igneum-genesis", V4_CLASS, &era, &V3_ALLOWED); + assert_eq!(via_era, v4, "the --era path and the chain path make the same v4 program"); + assert_eq!(via_era.generator, GENERATOR_VERSION_V4); + assert_eq!(generate_era("igneum-genesis", b"igneum-genesis", V3_CLASS, &era, &V3_ALLOWED), v3); + assert_eq!(generate_era("igneum-genesis", b"igneum-genesis", LoadClass::MX8.with_shadow(256, 13), &era, &V3_ALLOWED).generator, GENERATOR_VERSION_V3, "a measurement class stays generator 3"); + assert_eq!(ProgramClass::of_load_class(&v4.class), Some(ProgramClass::V4)); + assert_eq!(ProgramClass::of_load_class(&v3.class), Some(ProgramClass::V3)); + assert_eq!(ProgramClass::of_load_class(&LoadClass::V2), Some(ProgramClass::V2)); + assert_eq!(ProgramClass::of_load_class(&LoadClass::MX8.with_shadow(256, 13)), None); + assert_eq!(ProgramClass::of_load_class(&LoadClass::MX4), None); + // without an era (a template before the era is known) the bare V4_CLASS stands, generator 4 + let bare = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V4, None); + assert_eq!(bare.class, V4_CLASS); + assert_eq!(bare.generator, GENERATOR_VERSION_V4); + assert_eq!(bare.era_bytes, None); + // the block changes the hash over the same dataset; the same seed and era derive the same block again + let ds = crate::verify::DatasetSource::new("2026-10-06", crate::verify::DatasetMode::ClosedForm, 20); + let h3 = crate::verify::hash_warp(&v3, 0, &ds); + let h4 = crate::verify::hash_warp(&v4, 0, &ds); + assert_ne!(h3, h4); + let again = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era)); + assert_eq!(again.shadow, v4.shadow); + assert_eq!(crate::verify::hash_warp(&again, 0, &ds), h4); + // v2 and v3 are byte for byte what they were (the pinned packs are diffed in tests/packs.rs) + assert_eq!(V3_CLASS.shadow, None); + assert_eq!(generate("igneum-genesis").program_id(), 0xbcc1248b10cc90f2); + } + + /// Era layout: the draw is deterministic, within bounds, and the six test eras are pinned; an era program has + /// 16 loads with window draws in bounds and nothing drawn on ALU slots; the class names and program ids separate + /// the eras from each other and from every other class. + #[test] + fn era_draw_deterministic_and_bounded() { + let all = [1u8, 4, 16]; + for n in 0..200u64 { + let eb = EraParams::test_era_bytes(&format!("igneum-era-test/{n}")); + let e = era_draw(&eb, &all); + assert_eq!(e, era_draw(&eb, &all)); + assert_eq!(e.words, EraParams::stream_words(&eb)); + assert!(all.contains(&e.width_words)); + assert_eq!(e.stride_mul & 1, 1); + assert!((1..=31).contains(&e.stride_rot)); + assert!(e.layout().is_valid(), "{:?}", e.pos); + let b = e.width_words.trailing_zeros() as usize; + for i in 0..b { + assert_eq!(e.pos[i], i as u8, "the low positions are the identity for width {}", e.width_words); + } + // a pinned set consumes the draw and keeps the stride draws in step + let pinned = era_draw(&eb, &[1]); + assert_eq!(pinned.width_words, 1); + assert_eq!((pinned.stride_mul, pinned.stride_rot), (e.stride_mul, e.stride_rot)); + assert_ne!(era_draw(&EraParams::test_era_bytes(&format!("igneum-era-test/{}", n + 1)), &all).words, e.words); + } + // the six test eras pinned at 4 bytes (docs/plans/era-layout.md section 5): ERA_VECTORS + for (n, mul, rot, pos) in ERA_VECTORS { + let e = era_draw(&EraParams::test_era_bytes(&format!("igneum-era-test/{n}")), &V3_ALLOWED); + assert_eq!((e.width_words, e.stride_mul, e.stride_rot, e.pos), (1, mul, rot, pos), "era test seed {n}"); + } + // era programs + let mut ids = std::collections::HashSet::new(); + ids.insert(candidate("igneum-genesis", b"igneum-genesis", 0).program_id()); + ids.insert(candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::fixed(16, 16)).program_id()); + for n in 0..6u64 { + let eb = EraParams::test_era_bytes(&format!("igneum-era-test/{n}")); + let c = LoadClass::era(LoadClass::V2, &eb, &all); + let e = c.era.unwrap(); + assert_eq!(c.mix.iter().position(|&m| m == 100).map(|i| WIDTH_WORDS[i]), Some(e.width_words)); + let base = match e.width_words { + 1 => "w4", + 4 => "w16", + _ => "w64", + }; + assert_eq!(c.name(), format!("{base}-era{}", e.label())); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.class, c); + assert_eq!(p.loads_per_hash(), 128); + assert_eq!(p.bytes_per_hash(), 128 * 4 * e.width_words as usize); + assert_ne!(p.instrs[0].op, Op::Load); + for ins in &p.instrs { + assert_ne!(ins.dst, ins.src); + assert!((1..=31).contains(&ins.rot)); + if ins.op == Op::Load { + assert_eq!(ins.width, e.width_words); + assert!(ins.win <= 2 && (ins.off as u32) < (1u32 << ins.win), "win {} off {}", ins.win, ins.off); + } else { + assert_eq!((ins.width, ins.win, ins.off), (1, 0, 0)); + } + } + assert!(ids.insert(p.program_id()), "era {n}: program id collides"); + // the pinned form keeps the base mix and redraws the interleave for the base's widest width + let pinned = LoadClass::era(LoadClass::mixed([50, 35, 15]), &eb, &[1]); + assert_eq!(pinned.mix, [50, 35, 15]); + assert_eq!(pinned.era.unwrap().width_words, 16); + assert_eq!(pinned.era.unwrap().pos, [0, 1, 2, 3]); + assert!(ids.insert(candidate_class("igneum-genesis", b"igneum-genesis", 0, pinned).program_id())); + } + // the window draws take two more draws per instruction: the stream differs from the read-width class + let w64 = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::fixed(16, 16)); + let era0 = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::era(LoadClass::V2, &EraParams::test_era_bytes("igneum-era-test/0"), &all)); + assert_ne!(w64.instrs, era0.instrs); + } + + /// The era draws of the six test seeds at the 4-byte width: (test seed, stride multiplier, rotation, positions), + /// as `igneum-pow show --era igneum-era-test/` printed them on 5 October 2026 (docs/plans/era-layout.md 5). + const ERA_VECTORS: [(u64, u32, u32, [u8; 4]); 6] = [ + (0, 0x625e5ab3, 19, [0, 2, 10, 15]), + (1, 0xb2a9d70d, 6, [1, 3, 8, 13]), + (2, 0x2b4a5b97, 28, [1, 3, 4, 8]), + (3, 0x27ea7eff, 30, [2, 3, 8, 13]), + (4, 0x4d38603d, 10, [2, 9, 13, 15]), + (5, 0x03ac37ad, 22, [0, 2, 10, 13]), + ]; + + /// The six test eras generate accepted programs through the chain's class v3 path (the acceptance rule with the + /// era address mirror); generator 3, the era bytes recorded, the class V3_CLASS with the era inside. + #[test] + fn era_programs_are_accepted() { + for n in 0..6u64 { + let eb = EraParams::test_era_bytes(&format!("igneum-era-test/{n}")); + let p = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, Some(&eb)); + assert!(check(&p).is_ok(), "era {n}"); + assert!(p.attempt < MAX_ATTEMPTS); + assert_eq!(p.generator, GENERATOR_VERSION_V3); + assert_eq!(p.era_bytes.as_deref(), Some(&eb[..])); + assert_eq!(p, generate_era("igneum-genesis", b"igneum-genesis", V3_CLASS, &eb, &V3_ALLOWED)); + } + } + + /// Hot-table experiment: names and ids; a hot class with version 2 widths and no scratch takes the version 2 + /// stream, so its program is the version 2 program with k of the load slots turned into hot loads; the hot + /// table composes with a scratch class. + #[test] + fn hot_classes() { + assert_eq!(LoadClass::parse("hot64k4"), Some(LoadClass::hot(64, 4))); + assert_eq!(LoadClass::hot(64, 4).name(), "hot64k4"); + assert_eq!(LoadClass::parse("hot96k4").unwrap().name(), "hot96k4"); + assert_eq!(LoadClass::parse("hot0k4"), None); + assert_eq!(LoadClass::parse("hot64k17"), None); + assert_eq!(LoadClass::parse("hot64"), None); + // the added form: 16 + k slots, 16 dataset loads, no width roll, its own name and id + for (mb, k) in [(32u8, 4u8), (64, 4), (96, 4)] { + let c = LoadClass::parse(&format!("hot{mb}k{k}a")).unwrap(); + assert_eq!(c, LoadClass::hot_added(mb, k)); + assert_eq!(c.name(), format!("hot{mb}k{k}a")); + assert_eq!(c.load_slots as usize, 16 + k as usize); + assert_eq!(c.dataset_slots(), 16); + assert!(!c.takes_width_roll()); + assert_ne!(c, LoadClass::hot(mb, k)); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.loads_per_hash(), 128 + 8 * k as usize); + assert_eq!(p.hot_loads_per_hash(), 8 * k as usize); + assert_eq!(p.bytes_per_hash(), 512); + assert_eq!(p.items_per_warp(), 4096); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Hot).count(), k as usize); + assert!(p.instrs.iter().all(|i| i.width == 1)); + assert_ne!(p.program_id(), candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::hot(mb, k)).program_id()); + } + assert_eq!(LoadClass::parse("scr4k32+hot64k4a").unwrap().name(), "scr4k32+hot64k4a"); + assert_eq!(LoadClass::parse("scr4k32+hot64k4a").unwrap().dataset_slots(), 12); + assert_eq!(LoadClass::parse("hot64k48a"), None); + assert!(!LoadClass::hot(64, 4).is_v2()); + assert!(!LoadClass::hot(64, 4).takes_width_roll()); + assert!(LoadClass::V2.takes_width_roll() == false); + assert!(LoadClass::scratch(4, 32).takes_width_roll()); + assert!(LoadClass::fixed(4, 16).takes_width_roll()); + let v2 = candidate("igneum-genesis", b"igneum-genesis", 0); + let mut ids = std::collections::HashSet::new(); + ids.insert(v2.program_id()); + for (mb, k) in [(32u8, 4u8), (64, 4), (96, 4), (64, 2), (64, 8)] { + let c = LoadClass::hot(mb, k); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.class, c); + assert_eq!(p.loads_per_hash(), 128); + assert_eq!(p.hot_loads_per_hash(), 8 * k as usize); + assert_eq!(p.bytes_per_hash(), (16 - k as usize) * 8 * 4); + assert_eq!(p.hot_words(), crate::memhard::hot_words(mb as u32)); + assert!(ids.insert(p.program_id()), "hot{mb}k{k}: program id collides"); + // the version 2 program with k loads redirected: every other field identical, instruction by instruction + let mut hot = 0; + for (a, b) in p.instrs.iter().zip(v2.instrs.iter()) { + if a.op == Op::Hot { + hot += 1; + assert_eq!(b.op, Op::Load, "a hot slot is one of the version 2 load slots"); + assert_eq!((a.dst, a.src, a.src2, a.imm, a.imm2, a.rot, a.bit, a.mask, a.width), (b.dst, b.src, b.src2, b.imm, b.imm2, b.rot, b.bit, b.mask, b.width)); + } else { + assert_eq!(a, b); + } + } + assert_eq!(hot, k as usize); + } + // the same k at two sizes: the same instructions, different ids and tables + let a = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::hot(32, 4)); + let b = candidate_class("igneum-genesis", b"igneum-genesis", 0, LoadClass::hot(64, 4)); + assert_eq!(a.instrs, b.instrs); + assert_ne!(a.program_id(), b.program_id()); + // composition with a scratch class: scratch slots first, then hot, the rest dataset loads + let c = LoadClass::parse("scr4k32+hot64k4").unwrap(); + assert_eq!(c, LoadClass::scratch(4, 32).with_hot(64, 4)); + assert_eq!(c.name(), "scr4k32+hot64k4"); + assert!(c.takes_width_roll()); + let p = candidate_class("igneum-genesis", b"igneum-genesis", 0, c); + assert_eq!(p.scratch_ops_per_hash(), 32); + assert_eq!(p.hot_loads_per_hash(), 32); + assert_eq!(p.loads_per_hash(), 128); + assert_eq!(p.bytes_per_hash(), 8 * 8 * 4); + assert!(ids.insert(p.program_id())); + assert_eq!(LoadClass::parse("scr12k32+hot8k8"), None, "scratch and hot slots exceed the 16"); + assert_eq!(LoadClass::parse("w16+hot64k4").unwrap().name(), "w16+hot64k4"); + // the hot slots are a uniform subset of the load slots over a population + let mut position_sum = 0usize; + let mut n = 0usize; + for i in 0..200u32 { + let s = format!("igneum-hot-slots/{i}"); + let p = candidate_class(&s, s.as_bytes(), 0, LoadClass::hot(64, 4)); + let loads: Vec = p.instrs.iter().enumerate().filter(|(_, x)| x.op.is_load()).map(|(i, _)| i).collect(); + assert_eq!(loads.len(), 16); + for (rank, &i) in loads.iter().enumerate() { + if p.instrs[i].op == Op::Hot { + position_sum += rank; + n += 1; + } + } + } + assert_eq!(n, 800); + let mean_rank = position_sum as f64 / n as f64; + assert!((mean_rank - 7.5).abs() < 0.6, "hot slots sit anywhere among the 16 loads: mean rank {mean_rank}"); + } + + #[test] + fn generate_returns_an_accepted_program() { + let p = generate("igneum-genesis"); + assert!(check(&p).is_ok()); + assert_eq!(p.seed, attempt_words(b"igneum-genesis", p.attempt)); + let again = generate_from_seed_bytes("other label", b"igneum-genesis"); + assert_eq!(again.instrs, p.instrs); + assert_eq!(again.attempt, p.attempt); + } + + #[test] + fn shadow_class_leaves_the_base_program_and_class_v3_untouched() { + // Counter ASIC 3.0 item 8: the shadow is drawn after the 64 base instructions, so the base program, its + // attempt and its acceptance verdict are the class's without the shadow; v2 and v3 draw nothing. + // Since class v4 sub-version 2 (AP-F8-1) a 256-instruction block over MX8 is the class v4 shape and draws its + // load sources under the dataflow rule on every path, so the "untouched" property is shown on a 64-instruction + // block (a measurement class, no rule) and the 256-block is shown to obey the rule instead + let base = generate_class("igneum-genesis", LoadClass::MX8); + let sh64 = generate_class("igneum-genesis", LoadClass::MX8.with_shadow(64, 13)); + assert_eq!(sh64.instrs, base.instrs); + assert_eq!(sh64.attempt, base.attempt); + assert_eq!(sh64.shadow.len(), 64); + let sh = generate_class("igneum-genesis", LoadClass::MX8.with_shadow(256, 13)); + assert!(crate::accept::check_fresh_sources_v4(&sh).is_ok(), "a 256-block over MX8 draws under the class v4 source rule"); + assert!(base.shadow.is_empty() && !base.has_shadow() && base.shadow_instrs_per_hash() == 0); + assert_eq!(sh.shadow.len(), 256); + assert!(sh.has_shadow()); + assert!(sh.shadow.iter().all(|i| !i.op.is_load() && i.op != Op::WLoad && i.src != i.dst && (1..=31).contains(&i.rot) && i.width == 1)); + assert_eq!(sh.shadow_instrs_per_hash(), ITERATIONS * 256 * 13); + assert_eq!(sh.class.name(), "mx8+sh256x13"); + assert_eq!(LoadClass::parse("mx8+sh256x13"), Some(sh.class)); + assert_eq!(LoadClass::parse("v2+sh64x1"), Some(LoadClass::V2.with_shadow(64, 1))); + assert_eq!(LoadClass::parse("mx8+sh0x1"), None); + assert_eq!(LoadClass::parse("mx8+sh64x0"), None); + assert_ne!(sh.program_id(), base.program_id()); + assert_ne!(sh.program_id(), generate_class("igneum-genesis", LoadClass::MX8.with_shadow(256, 12)).program_id()); + assert_eq!(V3_CLASS.shadow, None); + assert_eq!(LoadClass::V2.shadow, None); + let v2 = generate("igneum-genesis"); + assert!(v2.shadow.is_empty()); + assert_eq!(v2.program_id(), 0xbcc1248b10cc90f2); + // the block changes the hash: the interpreter runs it (closed-form dataset, small, no cache needed) + let ds = crate::verify::DatasetSource::new("2026-10-06", crate::verify::DatasetMode::ClosedForm, 20); + let h0 = crate::verify::hash_warp(&base, 0, &ds); + let h1 = crate::verify::hash_warp(&sh, 0, &ds); + assert_ne!(h0, h1); + // the same seed and class derive the same block + let again = generate_class("igneum-genesis", LoadClass::MX8.with_shadow(256, 13)); + assert_eq!(again.shadow, sh.shadow); + assert_eq!(crate::verify::hash_warp(&again, 0, &ds), h1); + } + + /// Latency ladder (`docs/design/latency-ladder.md`), the known-failed case first: before the ladder a changed N + /// was a hard fork. Two nodes drawing class v4 at 27 and at 35 passes from the same seeds build the same base + /// program and the same shadow block, carry generator 4 on both and, under the id rule as it stood, the SAME id + /// (`program_id(4, seed, attempt)` reads no shadow size), yet their warps hash differently: every block of one is + /// invalid to the other and no pack line told them apart. After: the rung is a parameter of the chain's step, + /// rung 0 is `V4_CLASS` byte for byte, and a rung above carries its pass count in the id. + #[test] + fn latency_ladder_known_failed_a_changed_n_was_a_hard_fork_and_rungs_are_class_v4() { + let era = [7u8; 32]; + let today = generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era)); + let r0 = generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era), 0); + let r27 = generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era), V4_SHADOW_REPS); + assert_eq!(r0, today, "rung 0 is class v4 byte for byte"); + assert_eq!(r27, today, "27 passes is rung 0"); + let r1 = generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era), 35); + // the failed case, as it stood: the same base, the same block, the same seed words, generator 4 on both, one id + assert_eq!(r1.instrs, today.instrs, "the base program is the class's, draw for draw"); + assert_eq!(r1.shadow, today.shadow, "the block is the same draw; only the pass count moves"); + assert_eq!((r1.seed, r1.attempt, r1.generator, r1.era_bytes.clone()), (today.seed, today.attempt, today.generator, today.era_bytes.clone())); + assert_eq!(program_id(GENERATOR_VERSION_V4, &r1.seed, r1.attempt), program_id(GENERATOR_VERSION_V4, &today.seed, today.attempt), "the old rule gave both nodes one id"); + assert_eq!(r1.shadow_reps(), 35); + assert_eq!(r1.shadow_instrs_per_hash(), ITERATIONS * 256 * 35); + let ds = crate::verify::DatasetSource::new("2026-10-06", crate::verify::DatasetMode::ClosedForm, 20); + let h0 = crate::verify::hash_warp(&today, 0, &ds); + let h1 = crate::verify::hash_warp(&r1, 0, &ds); + assert_ne!(h0, h1, "a changed N is another hash: before the ladder, a hard fork"); + // after: the rung is in the id above rung 0; rung 0 keeps the id written before the ladder + assert_ne!(r1.program_id(), today.program_id(), "the rung is in the id"); + assert_eq!(r1.program_id(), program_id_class(GENERATOR_VERSION_V4, &r1.seed, r1.attempt, &r1.class)); + assert_eq!(today.program_id(), program_id(GENERATOR_VERSION_V4, &today.seed, today.attempt), "rung 0 keeps the v4 id"); + assert_eq!(r1.program_class(), ProgramClass::V4, "a rung is class v4"); + assert_eq!(v4_rung_reps(&r1.class), Some(35)); + assert_eq!(v4_rung_reps(&today.class), Some(27)); + assert_eq!(v4_rung_reps(&V4_CLASS), Some(27)); + assert_eq!(v4_rung_reps(&V3_CLASS), None); + assert_eq!(v4_rung_reps(&LoadClass::V2), None); + assert_eq!(v4_rung_reps(&LoadClass::MX8.with_shadow(64, 52)), None, "another block size is a measurement class, not a rung"); + assert_eq!(v4_class_at(0), V4_CLASS); + assert_eq!(v4_class_at(27), V4_CLASS); + assert_eq!(v4_class_at(35), LoadClass::parse("mx8+sh256x35").unwrap()); + assert_eq!(LoadClass { era: None, ..r1.class }, v4_class_at(35), "the era rides inside the rung's class"); + assert_eq!(r1.class.era, today.class.era, "the same era draw at every rung"); + assert!(check(&r1).is_ok(), "the acceptance rule reads the base program, which did not move"); + // every rung of the designed ladder is another program with its own id + let rungs = [27u16, 35, 53, 88, 173, 267]; + let ids: Vec = rungs.iter().map(|&r| generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V4, Some(&era), r).program_id()).collect(); + for i in 0..ids.len() { + for j in 0..i { + assert_ne!(ids[i], ids[j], "rungs {} and {} share an id", rungs[i], rungs[j]); + } + } + // the ops labels of the rungs, within 1 percent of the measured rungs of algorithm.md 5.3a + for (r, ops) in [(27u16, 102_100u64), (35, 132_100), (53, 199_600), (88, 330_700), (173, 649_400), (267, 1_001_600)] { + let got = v4_counted_ops(r); + assert!(got.abs_diff(ops) * 100 < ops, "reps {r}: {got} counted ops against the label {ops}"); + } + // other classes ignore the rung; class v4 without an era takes it + assert_eq!(generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V3, Some(&era), 35), generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V3, Some(&era))); + assert_eq!(generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V2, None, 35), generate_from_seed_bytes_program_class("igneum-genesis", b"igneum-genesis", ProgramClass::V2, None)); + let bare = generate_from_seed_bytes_program_class_shadow("igneum-genesis", b"igneum-genesis", ProgramClass::V4, None, 35); + assert_eq!((bare.generator, bare.shadow_reps(), bare.era_bytes.is_none()), (GENERATOR_VERSION_V4, 35, true)); + // the era stream consumes draws 8 and 9 after the seven it uses, so the seven are what they were: the pinned era + // packs of tests/packs.rs hold the values; here, the draw is a function of the bytes and the set alone + let e = era_draw(&era, &V3_ALLOWED); + assert_eq!(e, era_draw(&era, &V3_ALLOWED)); + assert_eq!(e.width_words, 1); + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/lib.rs b/tools/attack/adv-accept-v5/igneum-pow/src/lib.rs new file mode 100644 index 000000000..690ddb0bc --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/lib.rs @@ -0,0 +1,43 @@ +//! igneum-pow: the Igneum lottery hash, bit-exact with the Swift prototype in `proto-metal/main.swift`. +//! +//! The crate has five parts, each mirroring one section of the prototype: +//! +//! * [`seed`]: the 32-byte seed words from a string (FNV-1a 64, four salts) and the SplitMix64 stream. +//! * [`generator`]: the 64-instruction program drawn from a seed (version 2: 16 load slots, fresh sources). +//! * [`accept`]: the acceptance rule every candidate program must pass; a rejected candidate is replaced by the +//! next attempt of the same seed. +//! * [`memhard`]: the 256 MiB ChaCha12 cache and the 8-round dataset item derivation (`proto-metal/MEMHARD.md`). +//! * [`derive`]: the per-day item-derivation program of Counter ASIC 3.0 item 2 (a prototype behind a load class). +//! * [`verify`]: the 32-lane warp interpreter that computes the 64-bit hash on the CPU, deriving dataset +//! words on demand from the cache (or from the closed form, for the old packs). +//! * [`emit`]: the Metal, CUDA and OpenCL kernel text for a program, byte-identical to the Swift exporter. +//! * [`bind`]: the header binding (init words from the pre-PoW header hash and the nonce) and the 256-bit mapping. +//! +//! Nothing here depends on a crate outside the standard library. The integration points for the +//! rusty-kaspa fork (`docs/fork-map.md`) are [`verify::Epoch`], [`verify::Epoch::verify_block`] and +//! [`verify::Epoch::hash_warp`]; the node hands [`emit::Pack`] files to miners. + +// The lane loops are written index style on purpose so they read like the kernels they mirror, and the +// SplitMix64 `next` keeps the Swift name. +#![allow(clippy::needless_range_loop, clippy::should_implement_trait, clippy::large_enum_variant)] +#![allow(clippy::manual_slice_size_calculation, clippy::too_many_arguments)] + +pub mod accept; +pub mod bind; +pub mod blake2b; +pub mod derive; +pub mod emit; +pub mod generator; +pub mod memhard; +pub mod packcheck; +pub mod seed; +pub mod state; +pub mod verify; + +pub use bind::{block_init_words, day_bytes, pow256_from_lane, target64_from_le256}; +pub use accept::{check as accept_program, AcceptReport, Reject}; +pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, generate_from_seed_bytes_program_class_shadow, v4_class_at, v4_counted_ops, v4_rung_reps, v5_class_at, v5_rung_reps, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, GENERATOR_VERSION_V5, V3_CLASS, V4_CLASS, V4_SHADOW_INSTRS, V4_SHADOW_REPS, V5_CLASS, PROGRAM_SUBVERSION_V4}; +pub use memhard::{cache_log2_words, dataset_log2_words, days_since_genesis, growth_doublings, Cache, MemhardCpu, MixParams, Shape}; +pub use seed::{fnv1a64, seed_words, SplitMix64}; +pub use state::{StateLeaves, StateStream}; +pub use verify::{hash_warp, interpret_warp_init, verify_block, DatasetMode, DatasetSource, Epoch}; diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/main.rs b/tools/attack/adv-accept-v5/igneum-pow/src/main.rs new file mode 100644 index 000000000..95795cd5d --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/main.rs @@ -0,0 +1,502 @@ +//! igneum-pow CLI. +//! +//! igneum-pow bench --seed [--day ] [--closed-form] [--dataset-log2 28] [--warps 20] +//! igneum-pow export --seed --out [--day ] [--closed-form] [--dataset-log2 28] [--epoch-hex <64 hex> --day-hex ] +//! igneum-pow hash --seed --nonce [--day ] [--closed-form] [--dataset-log2 28] +//! igneum-pow hash-bound --seed --prehash <64 hex> --nonce [--day ] [--closed-form] [--dataset-log2 28] +//! [--epoch-hex <64 hex> --day-hex ] byte seeds instead of strings (Epoch::from_seed_bytes) +//! igneum-pow accept --seed [--epoch-hex <64 hex>] every candidate of the seed with its verdict (spec 01 section 1.4.6) +//! igneum-pow show --seed [--epoch-hex <64 hex>] the accepted program, one instruction per line +//! +//! Read-width experiment (5 October 2026, docs/plans/read-width.md): `--class v2|w4|w16|w64|w64x4|p4,p16,p64[xN]` +//! on every command selects the load class (default v2, the lottery hash). Nothing in a v2 run changes. +//! +//! Era layout (5 October 2026, docs/plans/era-layout.md): `--era igneum-era-test/` (a test era seed: the 32 bytes +//! of seed_words_from_bytes of the string, era index n) or `--era :<64 hex>` (the chain's 32-byte era seed E_n) +//! turns the chosen class into its era class; `--era-widths 4` (default: the read-width decision of 5 October 2026 +//! keeps v2's 4-byte load) is the allowed width set the era draws from; `4,16,64` lets the era draw the width. + +use igneum_pow::generator::{EraParams, GENERATOR_VERSION_V3}; + +use igneum_pow::emit::export_pack; +use igneum_pow::generator::{LoadClass, ProgramClass}; +use igneum_pow::memhard::{Cache, Shape}; +use igneum_pow::seed::day_key; +use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2}; +use std::time::Instant; + +struct Args { + cmd: String, + seed: String, + day: String, + out: Option, + closed_form: bool, + dataset_log2: u32, + warps: usize, + nonce: u64, + /// `hash-bound --count N`: N consecutive nonces from --nonce, one epoch build (gate G2, 5 October 2026). + count: u64, + prehash: String, + epoch_hex: Option, + day_hex: Option, + /// Class v5: the window's state stream file (`--state `, the IGSD1 format of `igneum_pow::state`), whose + /// leaves every item of the dataset is keyed by. + state: Option, + class: LoadClass, + /// Days since genesis for the cache growth rule of a class with `growth` (0: the genesis cache). + days: u64, + /// The program class (Counter ASIC 2.0 seam): v2 (default), v3 (V3_CLASS, generator 3) or v4 (V4_CLASS, generator 4). + program_class: Option, + /// The shadow pass count of a class v4 program at a rung of the latency ladder (`--shadow-reps`, 0 = the class's + /// own 27; `docs/design/latency-ladder.md`); read under `--program-class v4` only. + shadow_reps: u16, + /// The era seed bytes a class v3 chain program records (`--era-hex`). + era_hex: Option, + /// Era layout: `--era igneum-era-test/` or `--era :<64 hex>` composes the era class over `--class` with + /// generator 3 and the era bytes recorded (the measurement packs: v2's mixer under the era layout). + era: Option<(u64, Vec, String)>, + era_widths: Vec, +} + +/// `igneum-era-test/` or `:<64 hex>` -> (index, 32 era bytes, label). +fn parse_era(s: &str) -> Option<(u64, Vec, String)> { + if let Some(n) = s.strip_prefix("igneum-era-test/") { + let index: u64 = n.parse().ok()?; + return Some((index, EraParams::test_era_bytes(s).to_vec(), s.to_string())); + } + let (n, hex) = s.split_once(':')?; + let index: u64 = n.parse().ok()?; + let bytes = igneum_pow::bind::unhex(hex)?; + if bytes.len() != 32 { + return None; + } + Some((index, bytes, format!("igneum-era/{index}/{hex}"))) +} + +/// "4,16,64" (bytes) -> ascending words. +fn parse_widths(s: &str) -> Option> { + let mut v: Vec = s + .split(',') + .map(|x| match x.trim() { + "4" => Some(1u8), + "16" => Some(4), + "64" => Some(16), + _ => None, + }) + .collect::>>()?; + v.sort_unstable(); + v.dedup(); + if v.is_empty() { + return None; + } + Some(v) +} + +fn usage() -> ! { + eprintln!( + "igneum-pow --seed [--day 2026-10-03] [--closed-form] [--dataset-log2 28]\n\ + \x20 bench [--warps 20] fill the cache, then time the CPU verifier per 32-lane warp\n\ + \x20 export --out write the program pack (kernel.cu, kernel.cl, program.metal, memhard.h, ...)\n\ + \x20 hash --nonce print the 64-bit hash of one nonce (pack form, init words = seed words)\n\ + \x20 hash-bound --prehash <64 hex> --nonce [--count N] print the header-bound hash (bind.rs) of one 64-bit nonce (N consecutive: \"nonce hash\" lines)\n\ + \x20 accept every candidate of the seed (or --epoch-hex) with its acceptance verdict\n\ + \x20 show the accepted program, one instruction per line\n\ + \x20 --class C load class: v2 (default), mx4, mx8 (class v3: mixer x8, cache growth), dr (Counter ASIC 3.0 item 2: the per-day derivation program, dr736 = the x8-equivalent), w4, w16, w64, w64x4, p4,p16,p64[xN], m[g]\n\ + \x20 also: w4, w16, w64, w64x4, p4,p16,p64[xN], m[g], +shx (latency-shadow block of S ALU instructions x R passes per iteration, Counter ASIC 3.0 item 8)\n\ + \x20 --days N days since genesis for the cache growth rule of a class with it (default 0: the 2^26-word cache)\n\ + \x20 --program-class v2|v3|v4|v5 the program class of the seam (v3 = generator 3 on V3_CLASS, v4 = generator 4 on V4_CLASS = mx8+sh256x27, v5 = generator 5 on V5_CLASS = mx8+sh256x27+state, the chain's own derivation; --era-hex records the era seed)\n\ + \x20 --state class v5 (or any --class ...+state): the window's state stream (IGSD1 file, igneum-day-stream --out), whose leaves key every item\n\ + \x20 --shadow-reps N class v4 at a rung of the latency ladder: the shadow block's pass count (0 = the class's own 27; docs/design/latency-ladder.md), with --program-class v4\n\ + \x20 --era E era layout over --class: igneum-era-test/ or :<64 hex> (the 32-byte era seed E_n)\n\ + \x20 --era-widths 4[,16,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it)" + ); + std::process::exit(2) +} + +fn parse() -> Args { + let mut a = Args { + cmd: String::new(), + seed: "igneum-genesis".into(), + day: "2026-10-03".into(), + out: None, + closed_form: false, + state: None, + dataset_log2: DEFAULT_DATASET_LOG2, + warps: 20, + nonce: 0, + count: 1, + prehash: "00".repeat(32), + epoch_hex: None, + day_hex: None, + class: LoadClass::V2, + days: 0, + program_class: None, + era_hex: None, + shadow_reps: 0, + era: None, + era_widths: vec![1], + }; + let mut it = std::env::args().skip(1); + a.cmd = it.next().unwrap_or_else(|| usage()); + while let Some(k) = it.next() { + let mut val = || it.next().unwrap_or_else(|| usage()); + match k.as_str() { + "--seed" => a.seed = val(), + "--day" => a.day = val(), + "--out" => a.out = Some(val()), + "--closed-form" => a.closed_form = true, + "--dataset-log2" => a.dataset_log2 = val().parse().unwrap_or_else(|_| usage()), + "--warps" => a.warps = val().parse().unwrap_or_else(|_| usage()), + "--nonce" => a.nonce = val().parse().unwrap_or_else(|_| usage()), + "--count" => a.count = val().parse().unwrap_or_else(|_| usage()), + "--prehash" => a.prehash = val(), + "--epoch-hex" => a.epoch_hex = Some(val()), + "--day-hex" => a.day_hex = Some(val()), + "--class" => a.class = LoadClass::parse(&val()).unwrap_or_else(|| usage()), + "--days" => a.days = val().parse().unwrap_or_else(|_| usage()), + "--program-class" => a.program_class = Some(ProgramClass::parse(&val()).unwrap_or_else(|| usage())), + "--state" => a.state = Some(val()), + "--era-hex" => a.era_hex = Some(val()), + "--shadow-reps" => a.shadow_reps = val().parse().unwrap_or_else(|_| usage()), + "--era" => a.era = Some(parse_era(&val()).unwrap_or_else(|| usage())), + "--era-widths" => a.era_widths = parse_widths(&val()).unwrap_or_else(|| usage()), + _ => usage(), + } + } + if let Some((_, bytes, _)) = &a.era { + a.class = LoadClass::era(a.class, bytes, &a.era_widths); + } + a +} + +/// An era program is a class v3 program: generator 3 and the era bytes recorded (what the chain's +/// `Epoch::from_chain_seeds` does); the pack then carries IGNEUM_PROGRAM_CLASS "v3" and IGNEUM_ERA_SEED_HEX. +fn stamp_era(e: &mut Epoch, a: &Args) { + if let Some((_, bytes, _)) = &a.era { + // generator 4 on V4_CLASS (class v4, Counter ASIC 3.0), 3 on every other era program + e.program.generator = igneum_pow::generator::era_generator_of(&e.program.class); + e.program.era_bytes = Some(bytes.clone()); + } +} + +fn main() { + let a = parse(); + let mode = if a.closed_form { DatasetMode::ClosedForm } else { DatasetMode::MemoryHard }; + match a.cmd.as_str() { + "bench" => bench(&a, mode), + "export" => export(&a, mode), + "accept" => accept(&a), + "show" => show(&a), + "hash" => { + let (e, _) = epoch_of(&a, mode); + println!("{:016x}", e.hash(a.nonce as u32)); + } + "hash-bound" => { + let bytes = igneum_pow::bind::unhex(&a.prehash).unwrap_or_else(|| usage()); + let prehash: [u8; 32] = bytes.as_slice().try_into().unwrap_or_else(|_| usage()); + // --epoch-hex / --day-hex: the chain's byte seeds (Epoch::from_seed_bytes), as the worker protocol carries them + let (e, _) = epoch_of(&a, mode); + if a.count > 1 { + // gate G2 (5 October 2026, ported from branch ca2-era for the class v4 gate run): `--count N` prints + // "nonce hash" for N consecutive 64-bit nonces from --nonce, one epoch build, to re-hash a worker's + // found lines (the same prehash, target ff..ff) + for k in 0..a.count { + let n = a.nonce.wrapping_add(k); + println!("{n} {:016x}", e.hash_bound(&prehash, n)); + } + return; + } + let init = igneum_pow::bind::block_init_words(&prehash, a.nonce); + println!("init words {}", init.iter().map(|w| format!("{w:08x}")).collect::>().join(" ")); + println!("{:016x}", e.hash_bound(&prehash, a.nonce)); + } + _ => usage(), + } +} + +/// The epoch every command works on, and the day label for packs. `--epoch-hex`/`--day-hex` give the chain's byte +/// seeds (the day label then names the day bytes); else the string seed and day. `--program-class v3` draws the +/// program through the seam (generator 3 on `V3_CLASS`, the era bytes of `--era-hex` recorded) and sizes the +/// dataset for `--days` through `Epoch::chain_dataset_day`; `--class` is ignored under a program class (the class +/// names the load class). Closed-form mode is only for string seeds under the default class. +fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) { + let (mut e, label) = epoch_of_class(a, mode); + stamp_era(&mut e, a); + // class v5: the leaves of --state, built for the dataset's size; a state class without --state is refused here + // rather than at the first derivation + if e.program.class.state { + let Some(path) = &a.state else { + eprintln!("class {} keys every item by the window's state: give --state (igneum-day-stream --out)", e.program.class.name()); + std::process::exit(2); + }; + let stream = igneum_pow::state::StateStream::read_file(std::path::Path::new(path)).unwrap_or_else(|err| { + eprintln!("{err}"); + std::process::exit(2) + }); + let leaves = igneum_pow::state::StateLeaves::from_stream(&stream, e.dataset.log2_words); + eprintln!( + "state stream {}: chain block {} {}, root {}, {} records, {} leaves{}", + path, + stream.number, + igneum_pow::emit::hex_bytes(&stream.block), + igneum_pow::emit::hex_bytes(&stream.root), + stream.records.len(), + leaves.n(), + if leaves.sampled { " (sampled)" } else { "" } + ); + e.dataset = e.dataset.with_leaves(std::sync::Arc::new(leaves)); + } else if a.state.is_some() { + eprintln!("--state given for a class without state leaves ({}); use --program-class v5 or --class +state", e.program.class.name()); + std::process::exit(2); + } + (e, label) +} + +fn epoch_of_class(a: &Args, mode: DatasetMode) -> (Epoch, String) { + let era = a.era_hex.as_ref().map(|h| igneum_pow::bind::unhex(h).unwrap_or_else(|| usage())); + match (&a.epoch_hex, &a.day_hex) { + (Some(eh), Some(dh)) => { + let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage()); + let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage()); + let label = format!("igneum-epoch/{eh}/day/{dh}"); + let e = match a.program_class { + Some(pc) => Epoch { + program: Epoch::chain_program_shadow(&eb, era.as_deref(), pc, a.shadow_reps, &label), + dataset: Epoch::chain_dataset_day(&db, pc, a.days, a.dataset_log2), + }, + None => Epoch::from_seed_bytes_day(&eb, &db, &label, a.class, a.days, a.dataset_log2), + }; + (e, format!("bytes:{dh}")) + } + _ => { + let e = match a.program_class { + Some(pc) => { + let program = igneum_pow::generator::generate_from_seed_bytes_program_class_shadow(&a.seed, a.seed.as_bytes(), pc, era.as_deref(), a.shadow_reps); + let lc = pc.load_class(); + let shape = Shape::for_class_day(&lc, a.days); + let log2 = if lc.growth { igneum_pow::memhard::dataset_log2_words(a.dataset_log2, a.days) } else { a.dataset_log2 }; + Epoch { program, dataset: DatasetSource::new_shape(&a.day, mode, log2, shape) } + } + None => Epoch::new_class_day(&a.seed, &a.day, mode, a.dataset_log2, a.class, a.days), + }; + (e, a.day.clone()) + } + } +} + +fn bench(a: &Args, mode: DatasetMode) { + println!( + "igneum-pow bench: seed \"{}\", day \"{}\", dataset 2^{} words ({})", + a.seed, + a.day, + a.dataset_log2, + mode.name() + ); + let shape = Shape::for_class_day(&a.program_class.map(|pc| pc.load_class()).unwrap_or(a.class), a.days); + if mode == DatasetMode::MemoryHard { + // Time the cache fill on its own first (one core), then build the epoch (which fills it again). + let t0 = Instant::now(); + let c = Cache::fill_log2(day_key(&a.day), shape.cache_log2_words); + let fill_ms = t0.elapsed().as_secs_f64() * 1e3; + println!( + "cache: fill {fill_ms:.1} ms on one core (2^{} words, {} MiB, {} chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", + shape.cache_log2_words, + shape.cache_words() * 4 / (1 << 20), + c.segments(), + c.fnv1a64() + ); + drop(c); + } + if let Some(h) = a.class.hot { + // the hot table of the epoch on its own first (one core), then the epoch (which fills it again) + let t0 = Instant::now(); + let t = igneum_pow::memhard::HotTable::for_seed_bytes(a.seed.as_bytes(), h.mb as u32); + let fill_ms = t0.elapsed().as_secs_f64() * 1e3; + println!("hot table: {} MiB filled in {fill_ms:.1} ms on one core ({} chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", h.mb, igneum_pow::memhard::hot_segments(h.mb as u32), t.fnv1a64()); + } + let t0 = Instant::now(); + let (e, _) = epoch_of(a, mode); + let build_ms = t0.elapsed().as_secs_f64() * 1e3; + println!( + "program: class {}, {} loads/hash, {} bytes/hash, widths (1,4,16 words) {:?}, {} items/warp, mixer x{} ({} mixers/item), cache 2^{} words, op mix {}; epoch built in {build_ms:.1} ms", + e.program.class.name(), + e.program.loads_per_hash(), + e.program.bytes_per_hash(), + e.program.width_counts(), + e.program.items_per_warp(), + shape.mixer_mult, + shape.mixers_per_item(), + shape.cache_log2_words, + e.program.op_mix() + ); + if let Some(dp) = e.dataset.memhard().and_then(|m| m.params.derive.as_ref()) { + // Counter ASIC 3.0 item 2: the day's derivation program + println!( + "derivation program: {} instructions per round program, {} per item ({} GPU ops, {} chip ops, {} multiplies per item), attempt {}, fingerprint {:016x}, op mix {}", + dp.len, + dp.instr_count(), + dp.gpu_ops(), + dp.chip_ops(), + dp.muls(), + dp.attempt, + dp.fingerprint(), + dp.op_mix() + ); + } + if e.program.has_shadow() { + println!( + "shadow: {} instructions x {} passes per iteration, {} shadow instructions per hash, op mix {}", + e.program.shadow.len(), + e.program.shadow_reps(), + e.program.shadow_instrs_per_hash(), + e.program.shadow_op_mix() + ); + } + let bases = [0u32, 4096, 1_000_000]; + for &b in &bases { + let t = Instant::now(); + let r = e.interpret_warp(b); + let ms = t.elapsed().as_secs_f64() * 1e3; + println!( + "warp base {b}: single cold run {ms:.3} ms, {} items derived, lane0 {:016x} lane31 {:016x}", + r.items_derived, r.hashes[0], r.hashes[31] + ); + } + let n = a.warps.max(1); + let t = Instant::now(); + let mut sink = 0u64; + for i in 0..n { + let w = e.hash_warp((i as u32) * 32 + 65536); + sink ^= w[0]; + } + let avg = t.elapsed().as_secs_f64() * 1e3 / n as f64; + println!("CPU verify: {avg:.3} ms per 32-lane warp, avg of {n} (checksum {sink:016x})"); +} + +fn export(a: &Args, mode: DatasetMode) { + let out = a.out.clone().unwrap_or_else(|| usage()); + let t0 = Instant::now(); + // --epoch-hex / --day-hex: the chain's byte seeds; the day label then names the day bytes + let (e, day_label) = epoch_of(a, mode); + let build_ms = t0.elapsed().as_secs_f64() * 1e3; + println!("igneum-pow export {out}"); + println!( + "seed \"{}\", day \"{}\", dataset 2^{} words ({}), generator v{} attempt {} program id {:016x}, loads/hash {}; epoch built in {build_ms:.1} ms", + e.program.seed_string, + day_label, + e.dataset.log2_words, + e.dataset.mode().name(), + e.program.generator, + e.program.attempt, + e.program.program_id(), + e.program.loads_per_hash() + ); + println!("op mix: {}; class {}, {} bytes/hash, widths (1,4,16 words) {:?}", e.program.op_mix(), e.program.class.name(), e.program.bytes_per_hash(), e.program.width_counts()); + let source = format!("igneum-pow (Rust) CPU interpreter, generator v{}, {} dataset", e.program.generator, e.dataset.mode().name()); + let pack = export_pack(&e, &day_label, &source); + let dir = std::path::Path::new(&out); + if let Err(err) = pack.write_to(dir) { + eprintln!("FAIL: write error {err}"); + std::process::exit(1); + } + for (name, text) in &pack.files { + println!("wrote {}/{name} ({} bytes)", dir.display(), text.len()); + } + for (i, b) in pack.bases.iter().enumerate() { + println!("vector warp base {b}: lane0 {:016x} lane31 {:016x}", pack.outs[i][0], pack.outs[i][31]); + } + if e.dataset.mode() == DatasetMode::MemoryHard { + println!("cache FNV-1a 64 {:016x}", pack.vectors.cache_fnv); + } + println!("OVERALL: PASS (pack written)"); +} + +fn seed_bytes_of(a: &Args) -> (String, Vec) { + match &a.epoch_hex { + Some(eh) => (format!("igneum-epoch/{eh}"), igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage())), + None => (a.seed.clone(), a.seed.as_bytes().to_vec()), + } +} + +fn accept(a: &Args) { + let (label, bytes) = seed_bytes_of(a); + let t0 = Instant::now(); + let tries = igneum_pow::generator::attempts_class(&label, &bytes, a.class); + let ms = t0.elapsed().as_secs_f64() * 1e3; + for (p, verdict) in &tries { + match verdict { + Ok(()) => { + let r = igneum_pow::accept::check(p).unwrap(); + println!( + "attempt {}: ACCEPTED program id {:016x}, op mix {}, distinct {:.3} per hash, saturated {}, bias max {}", + p.attempt, + p.program_id(), + p.op_mix(), + r.distinct_mean(), + r.saturated, + r.bias_max + ); + } + Err(r) => println!("attempt {}: rejected, {r}", p.attempt), + } + } + println!("{} candidates in {ms:.2} ms", tries.len()); +} + +fn show(a: &Args) { + let (label, bytes) = seed_bytes_of(a); + // --program-class: the chain's own derivation (the era from --era-hex), as the node and the miner draw it; else + // the load class of --class, the era of --era composed over it + let era_hex_bytes = a.era_hex.as_deref().map(|h| igneum_pow::bind::unhex(h).unwrap_or_else(|| usage())); + let mut p = match a.program_class { + Some(pc) => igneum_pow::generator::generate_from_seed_bytes_program_class_shadow(&label, &bytes, pc, era_hex_bytes.as_deref(), a.shadow_reps), + None => igneum_pow::generator::generate_from_seed_bytes_class(&label, &bytes, a.class), + }; + if let (None, Some((_, eb, _))) = (a.program_class, &a.era) { + p.generator = igneum_pow::generator::era_generator_of(&p.class); + p.era_bytes = Some(eb.clone()); + } + println!( + "seed \"{}\" generator v{} class {} attempt {} program id {:016x} seed words {}", + p.seed_string, + p.generator, + p.class.name(), + p.attempt, + p.program_id(), + p.seed.iter().map(|w| format!("{w:08x}")).collect::>().join(" ") + ); + println!("op mix {} loads/hash {} bytes/hash {}", p.op_mix(), p.loads_per_hash(), p.bytes_per_hash()); + if let Some(e) = p.class.era { + println!( + "era {} ({}): width {} B, stride mul {:#010x} rot {}, interleave {:?}, windows (site:shrink:offset) {}", + e.label(), + a.era.as_ref().map(|x| x.2.as_str()).unwrap_or("?"), + e.width_words as u32 * 4, + e.stride_mul, + e.stride_rot, + e.pos, + p.instrs + .iter() + .enumerate() + .filter(|(_, i)| i.op == igneum_pow::generator::Op::Load) + .map(|(k, i)| format!("{k}:{}:{}", i.win, i.off)) + .collect::>() + .join(" ") + ); + } + for (k, i) in p.instrs.iter().enumerate() { + println!( + "{k:2}: {:5} dst={} src={} src2={} imm={:#010x} imm2={:#010x} rot={} bit={} mask={}{}", + i.op.name(), + i.dst, + i.src, + i.src2, + i.imm, + i.imm2, + i.rot, + i.bit, + i.mask, + if i.op == igneum_pow::generator::Op::Load && i.width > 1 { format!(" width={}B", i.width as u32 * 4) } else { String::new() } + ); + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/memhard.rs b/tools/attack/adv-accept-v5/igneum-pow/src/memhard.rs new file mode 100644 index 000000000..63a35ee50 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/memhard.rs @@ -0,0 +1,1060 @@ +//! The memory-hard dataset of `proto-metal/MEMHARD.md`: a 256 MiB cache of chained ChaCha12 blocks keyed +//! by the day key, and 64-byte dataset items derived by 8 dependent cache reads through a seed-parameterised +//! ARX-multiply mixer. The verifier holds the cache and never the dataset. +//! +//! All arithmetic is on u32 modulo 2^32. Rotations are by 1..31 at every call site. +//! +//! Counter ASIC 2.0 (5 October 2026, `docs/plans/mixer-x4.md`, behind the program class): the construction has a +//! [`Shape`], the mixer multiplier `m` and the cache size. Under `m` every mixer application of an item becomes +//! `m` applications with distinct round keys, the 8 dependent cache reads unchanged; the cache doubles when the +//! dataset doubles ([`growth_doublings`]). [`Shape::V2`] (`m = 1`, 2^26 words) is version 2 bit for bit. + +use crate::derive::{run_round, DeriveProgram, SoaState, DERIVE_REGS, SOA_LANES}; +use crate::generator::LoadClass; +use crate::seed::{day_key, fnv1a64_words, SplitMix64}; +use crate::state::StateLeaves; +use std::sync::Arc; + +pub const CACHE_LOG2_WORDS: usize = 26; +pub const CACHE_SEGMENT_LOG2_LINES: usize = 6; +/// 2^26 words = 256 MiB (the version 2 cache, and the v3 cache until the first dataset doubling). +pub const CACHE_WORDS: usize = 1 << CACHE_LOG2_WORDS; +/// 2^22 lines of 16 words. +pub const CACHE_LINES: usize = CACHE_WORDS >> 4; +/// 64 chained lines per segment. +pub const CACHE_LINES_PER_SEGMENT: usize = 1 << CACHE_SEGMENT_LOG2_LINES; +/// 2^16 independent segments. +pub const CACHE_SEGMENTS: usize = CACHE_LINES >> CACHE_SEGMENT_LOG2_LINES; +pub const CACHE_LINE_MASK: u32 = (CACHE_LINES - 1) as u32; +pub const ITEM_ROUNDS: usize = 8; +pub const CHACHA_ROUNDS: usize = 12; +/// The ChaCha constants "expand 32-byte k". +pub const CHACHA_SIGMA: [u32; 4] = [0x61707865, 0x3320646e, 0x79622d32, 0x6b206574]; +/// "Igne", "umMH". +pub const CACHE_TAG: [u32; 2] = [0x49676e65, 0x756d4d48]; +/// "Igne", "umHT": the chain tag of the hot table (hot-table experiment, `docs/plans/hot-table.md`). +pub const HOT_TAG: [u32; 2] = [0x49676e65, 0x756d4854]; +/// Domain tag of the hot key: `KH = seed_words_from_bytes("igneum-hot/" || epoch seed bytes)`. +pub const HOT_KEY_TAG: &[u8] = b"igneum-hot/"; +/// Words per MiB of hot table. +pub const HOT_WORDS_PER_MIB: u32 = 1 << 18; +/// Segments (64 chained lines of 16 words, 4 KiB) per MiB of hot table. +pub const HOT_SEGMENTS_PER_MIB: u32 = 256; + +/// The shape of the item derivation and of the cache: the mixer multiplier and the cache size. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct Shape { + /// Mixer applications per round (and after the last read): 1 under version 2, 4 under class v3. + pub mixer_mult: u32, + /// The cache is 2^cache_log2_words words (26 at genesis; 27 and 28 after the dataset doublings of 1.13.3). + pub cache_log2_words: u32, + /// Counter ASIC 3.0 item 2 (`crate::derive`): instructions per round program of the per-day derivation + /// program, which replaces the `mixer_mult` applications of `M_r` in every mixer slot when non-zero. 0 for + /// version 2 and class v3 (the fixed mixer). + pub derive_len: u32, + /// Class v5 (`docs/design/class-v5-stored-state.md`, 7 October 2026): the item derivation XORs the window's state + /// leaf `leaf(t)` into the 16 initial words before the first mixer (`crate::state`). `false` for every other class. + pub state: bool, +} + +impl Shape { + /// Version 2: one mixer application per round, a 2^26-word cache. + pub const V2: Shape = Shape { mixer_mult: 1, cache_log2_words: CACHE_LOG2_WORDS as u32, derive_len: 0, state: false }; + + /// The shape of a load class on day 0 of the chain (and on every day for a class without the growth rule). + pub fn for_class(class: &LoadClass) -> Shape { + Shape::for_class_day(class, 0) + } + + /// The shape of a load class on day `days_since_genesis` of the chain: the class's multiplier, and the cache + /// of [`cache_log2_words`] when the class has the growth rule, else 2^26 words. + pub fn for_class_day(class: &LoadClass, days_since_genesis: u64) -> Shape { + Shape { + mixer_mult: class.mixer_mult(), + cache_log2_words: if class.growth { cache_log2_words(days_since_genesis) } else { CACHE_LOG2_WORDS as u32 }, + derive_len: class.derive_len as u32, + state: class.state, + } + } + + pub fn is_v2(&self) -> bool { + *self == Shape::V2 + } + pub fn cache_words(&self) -> usize { + 1usize << self.cache_log2_words + } + pub fn cache_lines(&self) -> usize { + self.cache_words() >> 4 + } + pub fn cache_line_mask(&self) -> u32 { + (self.cache_lines() - 1) as u32 + } + pub fn cache_segments(&self) -> usize { + self.cache_lines() >> CACHE_SEGMENT_LOG2_LINES + } + pub fn log2_segments(&self) -> u32 { + self.cache_log2_words - 4 - CACHE_SEGMENT_LOG2_LINES as u32 + } + /// Mixer applications per item: `(ITEM_ROUNDS + 1) x m` (0 under a derivation program, which has no mixer). + pub fn mixers_per_item(&self) -> u32 { + if self.is_derived() { + 0 + } else { + (ITEM_ROUNDS as u32 + 1) * self.mixer_mult + } + } + /// Whether the item derivation is the per-day program of `crate::derive` (Counter ASIC 3.0 item 2). + pub fn is_derived(&self) -> bool { + self.derive_len != 0 + } + /// Instructions per item under a derivation program: `(ITEM_ROUNDS + 1) x derive_len`. + pub fn derive_instrs_per_item(&self) -> u32 { + (ITEM_ROUNDS as u32 + 1) * self.derive_len + } +} + +// -------------------------------------------------------------------------------------------------------------- +// Dataset growth, option C (spec 01 section 1.13.3 option (b) with the cache tied to the dataset's doublings) +// -------------------------------------------------------------------------------------------------------------- + +/// Days per year of the growth schedule: one year = 31,536,000 DAA seconds of 86,400 (spec 01 section 1.13.3). +pub const GROWTH_DAYS_PER_YEAR: u64 = 365; +/// The linear schedule of 1.13.3, 2 GiB at genesis plus 0.5 GiB per year, is `G x (1 + d / 1460)` for the genesis +/// size `G` and the day `d`: it doubles at day 1,460 (year 4), quadruples at day 4,380 (year 12), reaches 8x at +/// day 10,220 (year 28) and 16x at day 21,900 (year 60). +pub const GROWTH_DOUBLING_DAYS: u64 = 4 * GROWTH_DAYS_PER_YEAR; + +/// The number of dataset doublings reached by day `days_since_genesis` of the chain: `floor(log2(1 + d / 1460))`, +/// in integers (`1 + d / 1460` rounded down, then its integer log2, which equals the real log2's floor because a +/// power of two is an integer). 0 until day 1,459; 1 from day 1,460 (year 4); 2 from day 4,380 (year 12). +pub fn growth_doublings(days_since_genesis: u64) -> u32 { + (1 + days_since_genesis / GROWTH_DOUBLING_DAYS).ilog2() +} + +/// The cache size on day `d` under option C: 2^26 words doubled once per dataset doubling (256 MiB, 512 MiB from +/// year 4, 1 GiB from year 12). +pub fn cache_log2_words(days_since_genesis: u64) -> u32 { + CACHE_LOG2_WORDS as u32 + growth_doublings(days_since_genesis) +} + +/// The dataset size on day `d` under option (b) of 1.13.3: the genesis size (2^`genesis_log2_words` words: 28 for +/// the 1 GiB packs and the devnet, 29 for the designed 2 GiB) doubled once per doubling of the linear schedule. The +/// result is capped at 32 (the item index is 32 bits, spec 1.13.3). +pub fn dataset_log2_words(genesis_log2_words: u32, days_since_genesis: u64) -> u32 { + (genesis_log2_words + growth_doublings(days_since_genesis)).min(32) +} + +/// Days since genesis from two day indices of `bind::day_index` (the header's `timestamp_ms / 86,400,000`): the +/// day of the block and the day of the genesis header. A block before the genesis day (clock skew) is day 0. +pub fn days_since_genesis(day_index: u64, genesis_day_index: u64) -> u64 { + day_index.saturating_sub(genesis_day_index) +} + +#[inline(always)] +fn rotl(x: u32, n: u32) -> u32 { + x.rotate_left(n) +} + +/// The ChaCha quarter round with explicit rotations. +#[inline(always)] +fn qr(s: &mut [u32; 16], a: usize, b: usize, c: usize, d: usize, r1: u32, r2: u32, r3: u32, r4: u32) { + s[a] = s[a].wrapping_add(s[b]); + s[d] ^= s[a]; + s[d] = rotl(s[d], r1); + s[c] = s[c].wrapping_add(s[d]); + s[b] ^= s[c]; + s[b] = rotl(s[b], r2); + s[a] = s[a].wrapping_add(s[b]); + s[d] ^= s[a]; + s[d] = rotl(s[d], r3); + s[c] = s[c].wrapping_add(s[d]); + s[b] ^= s[c]; + s[b] = rotl(s[b], r4); +} + +/// `y = ChaCha12 core(x) + x`. Standard rotations 16, 12, 8, 7; column round then diagonal round, six times. +#[inline] +pub fn chacha_block(x: &[u32; 16]) -> [u32; 16] { + let mut y = *x; + for _ in 0..CHACHA_ROUNDS / 2 { + qr(&mut y, 0, 4, 8, 12, 16, 12, 8, 7); + qr(&mut y, 1, 5, 9, 13, 16, 12, 8, 7); + qr(&mut y, 2, 6, 10, 14, 16, 12, 8, 7); + qr(&mut y, 3, 7, 11, 15, 16, 12, 8, 7); + qr(&mut y, 0, 5, 10, 15, 16, 12, 8, 7); + qr(&mut y, 1, 6, 11, 12, 16, 12, 8, 7); + qr(&mut y, 2, 7, 8, 13, 16, 12, 8, 7); + qr(&mut y, 3, 4, 9, 14, 16, 12, 8, 7); + } + for i in 0..16 { + y[i] = y[i].wrapping_add(x[i]); + } + y +} + +/// Mixer parameters drawn from the day key, plus the [`Shape`] the mixer is applied under. Draw order: +/// ROT[0..7] (1..31), MUL[0..15] (odd), RC[0..15]. The shape is not drawn: it is the class's. Under a shape with +/// a derivation program (Counter ASIC 3.0 item 2) the same stream continues after the 40 draws with the program's +/// draws (`DeriveProgram::draw`); the mixer constants are still drawn (the item init uses MUL and RC) and the +/// mixer itself is not applied. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct MixParams { + pub key: [u32; 8], + pub rot: [u32; 8], + pub mul: [u32; 16], + pub rc: [u32; 16], + pub shape: Shape, + /// The per-day derivation program when `shape.derive_len != 0`, else `None`. + pub derive: Option, + /// Class v5: how many mixer blocks the AP-F4-1 rule redrew before this one (0 on every other class, and on most days). + pub redraws: u32, +} + +/// Class v5's mixer-draw rule (AP-F4-1): the NAF sum of the 16 multipliers at least this. +pub const MIXER_NAF_SUM_MIN: u32 = 163; +/// Class v5's mixer-draw rule: every multiplier's NAF weight at least this. +pub const MIXER_NAF_WORD_MIN: u32 = 4; +/// Class v5's mixer-draw rule: at least this many distinct rotation amounts among the eight. +pub const MIXER_DISTINCT_ROT_MIN: usize = 4; +/// Class v5's mixer-draw rule: redraws before the last block stands as drawn (never reached at 6.1e-4 per try). +pub const MIXER_REDRAW_CAP: u32 = 64; + +/// The non-adjacent-form weight of a 32-bit word: the number of non-zero digits of its NAF, the adders a +/// shift-and-add multiplier by that constant needs (the M1 metric of the weak-day census). +pub fn naf_weight(mut x: u64) -> u32 { + let mut w = 0; + while x != 0 { + if x & 1 == 1 { + w += 1; + // the digit is +1 or -1: take x to the nearest multiple of 4 + if x & 3 == 3 { + x += 1; + } else { + x -= 1; + } + } + x >>= 1; + } + w +} + +/// Whether a mixer block passes class v5's draw rule (AP-F4-1). +pub fn mixer_block_admissible(rot: &[u32; 8], mul: &[u32; 16]) -> bool { + let sum: u32 = mul.iter().map(|&m| naf_weight(m as u64)).sum(); + let words = mul.iter().all(|&m| naf_weight(m as u64) >= MIXER_NAF_WORD_MIN); + let mut distinct = rot.to_vec(); + distinct.sort_unstable(); + distinct.dedup(); + sum >= MIXER_NAF_SUM_MIN && words && distinct.len() >= MIXER_DISTINCT_ROT_MIN +} + +impl MixParams { + /// Version 2 shape. + pub fn new(key: [u32; 8]) -> Self { + Self::with_shape(key, Shape::V2) + } + pub fn with_shape(key: [u32; 8], shape: Shape) -> Self { + let mut rng = SplitMix64::new(key[0] as u64 | ((key[1] as u64) << 32)); + let mut rot = [0u32; 8]; + let mut mul = [0u32; 16]; + let mut rc = [0u32; 16]; + for r in rot.iter_mut() { + *r = 1 + rng.below(31) as u32; + } + for m in mul.iter_mut() { + *m = (rng.next() as u32) | 1; + } + for c in rc.iter_mut() { + *c = rng.next() as u32; + } + let mut redraws = 0u32; + if shape.state { + // Class v5 (docs/design/class-v5-stored-state.md section 11, AP-F4-1, the attack-pass lane's weak-day census): + // a mixer block whose multipliers are cheap on an adder datapath (NAF sum under 163, a word under NAF weight 4) + // or whose rotations repeat (under 4 distinct amounts) is redrawn from the next stream values, so no day is a + // weak day for a per-day LUT-recompute FPGA (the worst calendar day of the census, chain day 29,337, was 1.121x). + // About 6.1e-4 of days redraw. The derive program's draws (none under v5) come after, as before. + while !mixer_block_admissible(&rot, &mul) && redraws < MIXER_REDRAW_CAP { + for r in rot.iter_mut() { + *r = 1 + rng.below(31) as u32; + } + for m in mul.iter_mut() { + *m = (rng.next() as u32) | 1; + } + for c in rc.iter_mut() { + *c = rng.next() as u32; + } + redraws += 1; + } + } + let derive = if shape.is_derived() { Some(DeriveProgram::draw(&mut rng, shape.derive_len)) } else { None }; + Self { key, rot, mul, rc, shape, derive, redraws } + } + /// Parameters for a day string: the key is `seed_words("day/" + day)`. + pub fn for_day(day: &str) -> Self { + Self::new(day_key(day)) + } +} + +/// The dataset layout (era layout, `docs/plans/era-layout.md` section 1.2): word `w` of the dataset holds word +/// `j(w)` of item `t(w)`, where `j(w)` gathers the four bits of `w` at the ascending positions `pos` and `t(w)` is +/// `w` with those bits removed. [`Layout::LINEAR`] (`pos = [0, 1, 2, 3]`) is `dataset[w] = item(w >> 4)[w & 15]`, +/// the lottery hash's mapping. Every position is below 16, so the mapping is the same at every dataset size of +/// at least 2^16 words and an item keeps its value at every size. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct Layout { + pub pos: [u8; 4], +} + +impl Layout { + pub const LINEAR: Layout = Layout { pos: [0, 1, 2, 3] }; + + pub fn is_linear(&self) -> bool { + self.pos == [0, 1, 2, 3] + } + + /// Positions ascending, distinct, below 16. + pub fn is_valid(&self) -> bool { + self.pos.iter().all(|&p| p < 16) && (1..4).all(|i| self.pos[i] > self.pos[i - 1]) + } + + /// `(t, j)` of word index `w`. + #[inline(always)] + pub fn split(&self, w: u32) -> (u32, u32) { + if self.is_linear() { + return (w >> 4, w & 15); + } + let mut j = 0u32; + for (i, &p) in self.pos.iter().enumerate() { + j |= ((w >> p) & 1) << i; + } + // remove the highest position first so the lower ones stay where they are + let mut t = w; + for &p in self.pos.iter().rev() { + let p = p as u32; + let low = (1u32 << p) - 1; + t = (t & low) | ((t >> (p + 1)) << p); + } + (t, j) + } + + /// The word index of word `j` of item `t`: the inverse of [`Layout::split`]. + #[inline(always)] + pub fn join(&self, t: u32, j: u32) -> u32 { + if self.is_linear() { + return (t << 4) | (j & 15); + } + // insert the lowest position first: every later position counts the bit just inserted + let mut w = t; + for (i, &p) in self.pos.iter().enumerate() { + let p = p as u32; + let low = (1u32 << p) - 1; + w = ((w >> p) << (p + 1)) | (w & low) | (((j >> i) & 1) << p); + } + w + } +} + +impl Default for Layout { + fn default() -> Self { + Layout::LINEAR + } +} + +/// Round key `(r + 1) * 0x9E3779B9` mod 2^32. +#[inline(always)] +pub fn round_key(r: usize) -> u32 { + ((r + 1) as u32).wrapping_mul(0x9E3779B9) +} + +/// The round key of application `j` (0 <= j < m) of round `r` under multiplier `m`: `round_key(r * m + j)`. For +/// `m = 1` this is `round_key(r)`, version 2's key. +#[inline(always)] +pub fn round_key_mult(r: usize, j: usize, m: usize) -> u32 { + round_key(r * m + j) +} + +/// `M_r` on 16 words in place: per word `(s ^ (RC + rk)) * MUL`, then one ChaCha-shaped double round with +/// the four column rotations `ROT[0..3]` and the four diagonal rotations `ROT[4..7]`. +#[inline(always)] +pub fn mixer(s: &mut [u32; 16], rk: u32, mp: &MixParams) { + for i in 0..16 { + s[i] = (s[i] ^ mp.rc[i].wrapping_add(rk)).wrapping_mul(mp.mul[i]); + } + let r = &mp.rot; + qr(s, 0, 4, 8, 12, r[0], r[1], r[2], r[3]); + qr(s, 1, 5, 9, 13, r[0], r[1], r[2], r[3]); + qr(s, 2, 6, 10, 14, r[0], r[1], r[2], r[3]); + qr(s, 3, 7, 11, 15, r[0], r[1], r[2], r[3]); + qr(s, 0, 5, 10, 15, r[4], r[5], r[6], r[7]); + qr(s, 1, 6, 11, 12, r[4], r[5], r[6], r[7]); + qr(s, 2, 7, 8, 13, r[4], r[5], r[6], r[7]); + qr(s, 3, 4, 9, 14, r[4], r[5], r[6], r[7]); +} + +/// The cache for one day key: 2^log2_words words (256 MiB under version 2). +pub struct Cache { + pub key: [u32; 8], + pub log2_words: u32, + line_mask: u32, + words: Vec, +} + +impl Cache { + /// One segment: 64 chained lines written at `cache[seg * 1024 ..]`. + /// `in_j = prev XOR (sigma || K || seg || j || tag)`, `line_j = B(in_j)`, `prev_0 = 0`. + pub fn fill_segment(words: &mut [u32], seg: usize, key: &[u32; 8]) { + Self::fill_segment_tagged(words, seg, key, &CACHE_TAG) + } + + /// [`Cache::fill_segment`] with an explicit chain tag: [`CACHE_TAG`] for the cache, [`HOT_TAG`] for the hot + /// table of `docs/plans/hot-table.md` (the same chain, another key and tag). + pub fn fill_segment_tagged(words: &mut [u32], seg: usize, key: &[u32; 8], tag: &[u32; 2]) { + let base = (seg << CACHE_SEGMENT_LOG2_LINES) * 16; + let seg_words = &mut words[base..base + CACHE_LINES_PER_SEGMENT * 16]; + let mut prev = [0u32; 16]; + for (j, line) in seg_words.as_chunks_mut::<16>().0.iter_mut().enumerate() { + let mut x = [0u32; 16]; + x[..4].copy_from_slice(&CHACHA_SIGMA); + x[4..12].copy_from_slice(key); + x[12] = seg as u32; + x[13] = j as u32; + x[14] = tag[0]; + x[15] = tag[1]; + for i in 0..16 { + x[i] ^= prev[i]; + } + let y = chacha_block(&x); + line.copy_from_slice(&y); + prev = y; + } + } + + /// The version 2 cache on the calling thread: 65,536 chains of 64 ChaCha12 blocks, in segment order. + pub fn fill(key: [u32; 8]) -> Cache { + Self::fill_log2(key, CACHE_LOG2_WORDS as u32) + } + + /// A cache of 2^`log2_words` words (26, 27 or 28 under the growth rule; smaller sizes for tests): 2^(log2 - 10) + /// independent chains of 64 lines, the same chain function at every size, so a larger cache's first segments + /// are the smaller cache's segments word for word. + pub fn fill_log2(key: [u32; 8], log2_words: u32) -> Cache { + assert!((10..=30).contains(&log2_words), "cache log2 words must be in 10..=30"); + let shape = Shape { mixer_mult: 1, cache_log2_words: log2_words, derive_len: 0, state: false }; + let mut words = vec![0u32; shape.cache_words()]; + for seg in 0..shape.cache_segments() { + Self::fill_segment(&mut words, seg, &key); + } + Cache { key, log2_words, line_mask: shape.cache_line_mask(), words } + } + + pub fn for_day(day: &str) -> Cache { + Self::fill(day_key(day)) + } + + #[inline(always)] + pub fn words(&self) -> &[u32] { + &self.words + } + + pub fn lines(&self) -> usize { + self.words.len() >> 4 + } + pub fn line_mask(&self) -> u32 { + self.line_mask + } + pub fn segments(&self) -> usize { + self.lines() >> CACHE_SEGMENT_LOG2_LINES + } + + /// Cache line `a` (masked to the cache's lines) as 16 words. + #[inline(always)] + pub fn line(&self, a: u32) -> &[u32] { + let o = (a & self.line_mask) as usize * 16; + &self.words[o..o + 16] + } + + /// [`Cache::line`] with the mask as a constant (the verifier's hot path, see [`derive_items`]). + #[inline(always)] + pub fn line_const(&self, a: u32) -> &[u32] { + let o = (a & LINE_MASK) as usize * 16; + &self.words[o..o + 16] + } + + /// FNV-1a 64 over the cache as little-endian bytes (what `vectors.h` carries as `IGNEUM_CACHE_FNV64`). + pub fn fnv1a64(&self) -> u64 { + fnv1a64_words(&self.words) + } +} + +/// The hot key of an epoch: `seed_words_from_bytes("igneum-hot/" || seed_bytes)`, `seed_bytes` the program seed +/// bytes before any attempt suffix, so every attempt of one epoch shares one table. +pub fn hot_key(seed_bytes: &[u8]) -> [u32; 8] { + let mut b = Vec::with_capacity(HOT_KEY_TAG.len() + seed_bytes.len()); + b.extend_from_slice(HOT_KEY_TAG); + b.extend_from_slice(seed_bytes); + crate::seed::seed_words_from_bytes(&b) +} + +/// Words of a hot table of `mb` MiB. +pub fn hot_words(mb: u32) -> u32 { + mb * HOT_WORDS_PER_MIB +} + +/// Segments of a hot table of `mb` MiB. +pub fn hot_segments(mb: u32) -> u32 { + mb * HOT_SEGMENTS_PER_MIB +} + +/// The hot index of a source word: `mulhi(src, words)`, the high 32 bits of the 64-bit product, in `[0, words)` +/// for any table size (the multiply-shift range reduction of spec 01 section 1.13.3). +#[inline(always)] +pub fn hot_index(src: u32, words: u32) -> u32 { + ((src as u64 * words as u64) >> 32) as u32 +} + +/// The hot table `H` of one epoch (hot-table experiment): `mb` MiB of chained ChaCha12 lines under the hot key, +/// read by the hot load slots as `dst ^= H[hot_index(src, words)]`. The verifier holds it beside the cache. +pub struct HotTable { + pub key: [u32; 8], + pub mb: u32, + words: Vec, +} + +impl HotTable { + /// Fill `mb` MiB under `key` on the calling thread. + pub fn fill(key: [u32; 8], mb: u32) -> HotTable { + assert!(mb >= 1 && mb <= 4096, "hot table size in MiB out of range"); + let n = hot_words(mb) as usize; + let mut words = vec![0u32; n]; + for seg in 0..hot_segments(mb) as usize { + Cache::fill_segment_tagged(&mut words, seg, &key, &HOT_TAG); + } + HotTable { key, mb, words } + } + + /// The table of the epoch whose program seed bytes are `seed_bytes`. + pub fn for_seed_bytes(seed_bytes: &[u8], mb: u32) -> HotTable { + Self::fill(hot_key(seed_bytes), mb) + } + + #[inline(always)] + pub fn n_words(&self) -> u32 { + self.words.len() as u32 + } + + /// `H[i]`. + #[inline(always)] + pub fn at(&self, i: u32) -> u32 { + self.words[i as usize] + } + + /// `H[hot_index(src, words)]`: what a hot load reads for source word `src`. + #[inline(always)] + pub fn word(&self, src: u32) -> u32 { + self.words[hot_index(src, self.n_words()) as usize] + } + + #[inline(always)] + pub fn words(&self) -> &[u32] { + &self.words + } + + /// FNV-1a 64 over the table as little-endian bytes (what `vectors.h` carries as `IGNEUM_HOT_FNV64`). + pub fn fnv1a64(&self) -> u64 { + fnv1a64_words(&self.words) + } +} + +/// Derive `ts.len()` items into `out`, all chains interleaved round by round so the cache-line misses of +/// independent items overlap in the memory system (`deriveItems` in the Swift). Under multiplier `m` +/// (`mp.shape.mixer_mult`) round `r` applies `M` with keys `round_key(r m + j)` for `j = 0 .. m - 1` before its +/// one cache read; the final mixer applies `M` with keys `round_key(8 m + j)`. `m = 1` is version 2. +pub fn derive_items(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]]) { + derive_items_leaves(ts, mp, cache, None, out) +} + +/// [`derive_items`] with the state leaves of class v5 (`docs/design/class-v5-stored-state.md` section 2): under a +/// shape with `state`, `leaf(t)` is XORed into the 16 initial words of item `t` before the first mixer, and the leaves +/// are required; under any other shape they must be absent. A mismatch is a programming error and panics: a dataset +/// built without the state it needs would be wrong on every item, which is the class's point. +pub fn derive_items_leaves(ts: &[u32], mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) { + match (mp.shape.state, leaves) { + (true, None) => panic!("class v5 item derivation needs the window's state leaves and was given none"), + (false, Some(_)) => panic!("state leaves given to an item derivation whose shape has no state"), + _ => {} + } + if let Some(prog) = &mp.derive { + return derive_items_program(ts, mp, prog, cache, leaves, out); + } + // The item loop lives in its own function, one instance per cache size the growth rule can reach with the line + // mask a constant, never inlined into the callers. Inlined into `MemhardCpu::fetch` it ran at 1.33 ms per unit + // against 0.61 out of line (the version 2 verifier, bisected on one core under the measure lock, 5 October 2026, + // `docs/plans/mixer-x4.md` section 6.6: the constant mask alone, or the mask hoisted into a local, or the + // constant with the loop still inlined, all stayed at 1.33; the out-of-line instances read 0.60 to 0.62). Any + // other cache size (tests) takes the instance with the run-time mask. + match cache.log2_words { + 26 => derive_items_mask::<{ (1u32 << 22) - 1 }>(ts, mp, cache, leaves, out), + 27 => derive_items_mask::<{ (1u32 << 23) - 1 }>(ts, mp, cache, leaves, out), + 28 => derive_items_mask::<{ (1u32 << 24) - 1 }>(ts, mp, cache, leaves, out), + 29 => derive_items_mask::<{ (1u32 << 25) - 1 }>(ts, mp, cache, leaves, out), + 30 => derive_items_mask::<{ (1u32 << 26) - 1 }>(ts, mp, cache, leaves, out), + _ => derive_items_mask::<0>(ts, mp, cache, leaves, out), + } +} + +/// [`derive_items`] with the cache line mask as a constant (`LINE_MASK = 0`: the cache's own run-time mask). Kept +/// out of line on purpose (see [`derive_items`]). +#[inline(never)] +fn derive_items_mask(ts: &[u32], mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) { + let n = ts.len(); + debug_assert!(out.len() >= n); + debug_assert!(LINE_MASK == 0 || LINE_MASK == cache.line_mask); + let m = mp.shape.mixer_mult as usize; + for k in 0..n { + let s = &mut out[k]; + let t = ts[k]; + s[..8].copy_from_slice(&mp.key); + for i in 0..8 { + s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); + } + if let Some(l) = leaves { + // class v5: the window's state leaf of item t, before the first mixer + let leaf = l.leaf(t); + for i in 0..16 { + s[i] ^= leaf[i]; + } + } + } + for r in 0..ITEM_ROUNDS { + for j in 0..m { + let rk = round_key_mult(r, j, m); + for s in out[..n].iter_mut() { + mixer(s, rk, mp); + } + } + for s in out[..n].iter_mut() { + let line = if LINE_MASK != 0 { cache.line_const::(s[0]) } else { cache.line(s[0]) }; + for i in 0..16 { + s[i] ^= line[i]; + } + } + } + for j in 0..m { + let rk = round_key_mult(ITEM_ROUNDS, j, m); + for s in out[..n].iter_mut() { + mixer(s, rk, mp); + } + } +} + +/// [`derive_items`] under a per-day derivation program (Counter ASIC 3.0 item 2, `crate::derive`): the same +/// init and the same 8 dependent cache reads, with round program `r` in place of the `m` mixer applications of +/// round `r` and program 8 in place of the final applications. The states are kept word-major +/// (`st[reg][lane]`) so every instruction runs across the batch's items in one vectorised loop and the +/// interpreter's dispatch is paid once per instruction per batch of up to 32 items, not once per item. The +/// cache reads of the batch are issued together, as in the fixed-mixer loop, so the 8 dependent misses of +/// independent items overlap in the memory system. +#[inline(never)] +pub fn derive_items_program(ts: &[u32], mp: &MixParams, prog: &DeriveProgram, cache: &Cache, leaves: Option<&StateLeaves>, out: &mut [[u32; 16]]) { + let n = ts.len(); + debug_assert!(out.len() >= n && n <= SOA_LANES); + assert_eq!(prog.rounds.len(), ITEM_ROUNDS + 1); + let mut st: SoaState = [[0u32; SOA_LANES]; DERIVE_REGS]; + for k in 0..n { + let t = ts[k]; + for i in 0..8 { + st[i][k] = mp.key[i]; + st[8 + i][k] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); + } + if let Some(l) = leaves { + let leaf = l.leaf(t); + for i in 0..16 { + st[i][k] ^= leaf[i]; + } + } + } + let mask = cache.line_mask(); + for r in 0..ITEM_ROUNDS { + run_round(&prog.rounds[r], &mut st); + for k in 0..n { + let line = cache.line(st[0][k] & mask); + for i in 0..16 { + st[i][k] ^= line[i]; + } + } + } + run_round(&prog.rounds[ITEM_ROUNDS], &mut st); + for k in 0..n { + for i in 0..16 { + out[k][i] = st[i][k]; + } + } +} + +/// One dataset item, 16 words. +pub fn derive_item(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] { + derive_item_leaves(t, mp, cache, None) +} + +/// [`derive_item`] with the state leaves of class v5. +pub fn derive_item_leaves(t: u32, mp: &MixParams, cache: &Cache, leaves: Option<&StateLeaves>) -> [u32; 16] { + let mut out = [[0u32; 16]; 1]; + derive_items_leaves(&[t], mp, cache, leaves, &mut out); + out[0] +} + +/// The CPU verifier's view of the memory-hard dataset: the mixer parameters (with the shape), the cache (shared, so +/// a class v5 window refresh keeps the day's 256 MiB and swaps the leaves) and, under class v5, the window's leaves. +pub struct MemhardCpu { + pub params: MixParams, + pub cache: Arc, + pub leaves: Option>, +} + +/// Largest batch `MemhardCpu::fetch` accepts (two warps). +pub const FETCH_MAX: usize = 64; + +impl MemhardCpu { + /// Version 2 shape. + pub fn new(key: [u32; 8]) -> Self { + Self::with_shape(key, Shape::V2) + } + pub fn with_shape(key: [u32; 8], shape: Shape) -> Self { + Self { params: MixParams::with_shape(key, shape), cache: Arc::new(Cache::fill_log2(key, shape.cache_log2_words)), leaves: None } + } + pub fn for_day(day: &str) -> Self { + Self::new(day_key(day)) + } + pub fn shape(&self) -> Shape { + self.params.shape + } + /// This view with the window's state leaves (class v5). The shape must have `state`. + pub fn with_leaves(mut self, leaves: Arc) -> Self { + assert!(self.params.shape.state, "state leaves on a shape without state"); + self.leaves = Some(leaves); + self + } + /// A view of the same day (the same cache, shared) with other leaves: the class v5 window refresh. + pub fn refreshed(&self, leaves: Arc) -> Self { + assert!(self.params.shape.state, "state leaves on a shape without state"); + Self { params: self.params.clone(), cache: self.cache.clone(), leaves: Some(leaves) } + } + /// `dataset[w] = item(w >> 4)[w & 15]` (the linear layout). + pub fn word(&self, w: u32) -> u32 { + self.word_at(Layout::LINEAR, w) + } + /// `dataset[w] = item(t(w))[j(w)]` under `layout` (era layout; the layout is the program's, the cache the + /// day's, so one cache serves every era of a day). + pub fn word_at(&self, layout: Layout, w: u32) -> u32 { + let (t, j) = layout.split(w); + derive_item_leaves(t, &self.params, &self.cache, self.leaves.as_deref())[j as usize] + } + /// `out[k] = dataset[idx[k]]` for every k, `idx.len() <= FETCH_MAX`. Equal items are derived once. + /// Returns the number of distinct items derived. + pub fn fetch(&self, idx: &[u32], out: &mut [u32], layout: Layout) -> usize { + let n = idx.len(); + assert!(n <= FETCH_MAX && out.len() >= n); + let mut uniq = [0u32; FETCH_MAX]; + let mut slot = [0u8; FETCH_MAX]; + let mut word = [0u8; FETCH_MAX]; + let mut u = 0usize; + for k in 0..n { + let (t, j) = layout.split(idx[k]); + word[k] = j as u8; + let found = uniq[..u].iter().position(|&x| x == t); + let j = match found { + Some(j) => j, + None => { + uniq[u] = t; + u += 1; + u - 1 + } + }; + slot[k] = j as u8; + } + let mut items = [[0u32; 16]; FETCH_MAX]; + derive_items_leaves(&uniq[..u], &self.params, &self.cache, self.leaves.as_deref(), &mut items); + for k in 0..n { + out[k] = items[slot[k] as usize][word[k] as usize]; + } + u + } + /// `out[k][j] = dataset[base[k] + j]` for `j < width` (read-width experiment): `base[k]` is aligned to `width` + /// words and the layout's low `log2(width)` positions are the identity, so every lane's words lie in one item + /// at consecutive word offsets, derived once per distinct item. Returns the distinct items. + pub fn fetch_wide(&self, base: &[u32], width: usize, out: &mut [[u32; 16]], layout: Layout) -> usize { + let n = base.len(); + assert!(n <= FETCH_MAX && out.len() >= n && width <= 16); + debug_assert!((0..width.trailing_zeros() as usize).all(|i| layout.pos[i] == i as u8), "a wide load needs the identity on its low positions"); + let mut uniq = [0u32; FETCH_MAX]; + let mut slot = [0u8; FETCH_MAX]; + let mut word = [0u8; FETCH_MAX]; + let mut u = 0usize; + for k in 0..n { + let (t, j0) = layout.split(base[k]); + word[k] = j0 as u8; + let j = match uniq[..u].iter().position(|&x| x == t) { + Some(j) => j, + None => { + uniq[u] = t; + u += 1; + u - 1 + } + }; + slot[k] = j as u8; + } + let mut items = [[0u32; 16]; FETCH_MAX]; + derive_items_leaves(&uniq[..u], &self.params, &self.cache, self.leaves.as_deref(), &mut items); + for k in 0..n { + let o = word[k] as usize; + out[k][..width].copy_from_slice(&items[slot[k] as usize][o..o + width]); + } + u + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Class v5's mixer-draw rule (AP-F4-1), the known-failed case first: a block of cheap multipliers (NAF sum under + /// 163) or repeated rotations is inadmissible; a scan of day keys finds days the rule redraws (the census's 6.1e-4), + /// every v5 block passes after the draw, and the v4 constants of the same keys never move. + #[test] + fn class_v5_mixer_draw_rule() { + assert_eq!(naf_weight(0), 0); + assert_eq!(naf_weight(1), 1); + assert_eq!(naf_weight(3), 2, "11 = 100 - 1"); + assert_eq!(naf_weight(7), 2, "111 = 1000 - 1"); + assert_eq!(naf_weight(0xffff_ffff), 2); + assert_eq!(naf_weight(0b1010_1010), 4); + let good_rot = [1u32, 5, 9, 13, 17, 21, 25, 29]; + let cheap = [0x8000_0001u32; 16]; + assert!(!mixer_block_admissible(&good_rot, &cheap), "the known-failed case: 16 two-adder multipliers"); + let dense = [0xaaaa_aaabu32; 16]; + assert!(mixer_block_admissible(&good_rot, &dense)); + assert!(!mixer_block_admissible(&[7u32; 8], &dense), "one rotation amount"); + assert!(!mixer_block_admissible(&[1u32, 2, 3, 3, 3, 3, 3, 3], &dense), "three distinct amounts"); + let v5 = Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: true }; + let v4 = Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: false }; + let mut redrawn = 0; + let mut scanned = 0; + for d in 0..60_000u64 { + let key = crate::seed::seed_words_from_bytes(&crate::bind::day_bytes(20_000 + d)); + let a = MixParams::with_shape(key, v5); + assert!(mixer_block_admissible(&a.rot, &a.mul), "day {d}: a v5 block fails the rule after the draw"); + if a.redraws > 0 { + redrawn += 1; + let b = MixParams::with_shape(key, v4); + assert_eq!(b.redraws, 0, "v4 never redraws"); + assert_ne!((a.rot, a.mul), (b.rot, b.mul), "day {d}: v5 redrew, v4 kept the block"); + assert!(!mixer_block_admissible(&b.rot, &b.mul), "day {d}: the v4 block was the inadmissible one"); + } else { + let b = MixParams::with_shape(key, v4); + assert_eq!((a.rot, a.mul, a.rc), (b.rot, b.mul, b.rc), "day {d}: an admissible day is byte for byte v4's"); + } + scanned += 1; + if redrawn >= 3 && scanned >= 2_000 { + break; + } + } + assert!(redrawn >= 1, "no redraw in {scanned} days (the census says about 6.1e-4 per day)"); + } + + #[test] + fn mix_params_for_day() { + // MEMHARD.md section 1.4 and the igneum-genesis-mh pack. + let mp = MixParams::for_day("2026-10-03"); + assert_eq!(mp.rot, [20, 20, 19, 4, 26, 3, 3, 27]); + assert_eq!(mp.mul[0], 0x42146205); + assert_eq!(mp.mul[15], 0x99cfb423); + assert_eq!(mp.rc[0], 0xbab68293); + assert_eq!(mp.rc[15], 0x31b49ee2); + assert!(mp.mul.iter().all(|m| m & 1 == 1)); + assert_eq!(mp.shape, Shape::V2); + } + + /// Era layout: split and join are inverse, the linear layout is today's mapping, and an interleaved layout + /// keeps every position below 16 so the mapping is the same at every size of at least 2^16 words. + #[test] + fn layout_split_join() { + let lin = Layout::LINEAR; + assert!(lin.is_linear() && lin.is_valid()); + for w in [0u32, 1, 15, 16, 17, 0x0fff_ffff, 0xffff_ffff] { + assert_eq!(lin.split(w), (w >> 4, w & 15)); + assert_eq!(lin.join(w >> 4, w & 15), w); + } + let l = Layout { pos: [0, 1, 7, 12] }; + assert!(!l.is_linear() && l.is_valid()); + for w in [0u32, 1, 2, 3, 4, 127, 128, 129, 4095, 4096, 0x0fff_ffff, 0x1234_5678, 0xffff_ffff] { + let (t, j) = l.split(w); + assert!(j < 16); + assert_eq!(l.join(t, j), w, "w {w:#x}"); + } + // bits: j0 = bit 0, j1 = bit 1, j2 = bit 7, j3 = bit 12; t = the other 28 bits in order + assert_eq!(l.split(0b1_0000_0000_0000), (0, 8)); + assert_eq!(l.split(1 << 7), (0, 4)); + assert_eq!(l.split(0b100), (1, 0)); + // every t in 0..2^(D-4) appears exactly once among w < 2^D (D = 16), with every j + let mut seen = vec![0u32; 1 << 12]; + for w in 0..(1u32 << 16) { + let (t, j) = l.split(w); + seen[t as usize] |= 1 << j; + } + assert!(seen.iter().all(|&s| s == 0xffff)); + assert!(!Layout { pos: [0, 1, 1, 5] }.is_valid()); + assert!(!Layout { pos: [0, 1, 2, 16] }.is_valid()); + assert!(!Layout { pos: [1, 0, 2, 3] }.is_valid()); + } + + #[test] + fn chacha_block_is_a_permutation_plus_feedforward() { + let x = [1u32; 16]; + let y = chacha_block(&x); + assert_ne!(x, y); + let z = chacha_block(&x); + assert_eq!(y, z); + } + + /// Hot-table experiment: the genesis epoch's table (seed bytes "igneum-genesis") as the hot packs carry it + /// (`proto-cuda/packs-ca2-hot/hot32k4/vectors.json`: hot_head, hot_fnv1a64; the head is the same at every size, + /// a larger table is more segments). The index mapping stays inside the table for any size. + #[test] + fn hot_table_fill_vector_and_index() { + assert_ne!(HOT_TAG, CACHE_TAG); + let h = HotTable::for_seed_bytes(b"igneum-genesis", 32); + assert_eq!(h.n_words(), 1 << 23); + assert_eq!(hot_segments(32), 8192); + assert_eq!( + &h.words()[..16], + &[ + 0x8068cc73, 0x6036ebf9, 0xb604cd25, 0x8ffb840e, 0xc54074a2, 0x285c0695, 0x77512425, 0xc26a58a7, + 0x72c88757, 0xc10fca78, 0x513825dd, 0x30d6ccc8, 0x9a05e7cf, 0xb9533f50, 0x4bac3ba0, 0xa5c19528 + ] + ); + assert_eq!(h.fnv1a64(), 0xc1767ba3ef02719f, "hot32k4 pack, hot_fnv1a64"); + assert_eq!(h.key, hot_key(b"igneum-genesis")); + assert_ne!(h.key, day_key("2026-10-03")); + // a different seed, a different table; the same seed under the cache tag is not the hot table + assert_ne!(HotTable::for_seed_bytes(b"igneum-genesis\x01\x00\x00\x00", 1).words()[..16], h.words()[..16]); + let mut under_cache_tag = vec![0u32; 1024]; + Cache::fill_segment(&mut under_cache_tag, 0, &h.key); + assert_ne!(&under_cache_tag[..16], &h.words()[..16]); + for words in [hot_words(32), hot_words(64), hot_words(96)] { + assert_eq!(hot_index(0, words), 0); + assert!(hot_index(u32::MAX, words) < words); + assert_eq!(hot_index(u32::MAX, words), words - 1); + assert!(hot_index(0x8000_0000, words) == words / 2); + } + assert_eq!(hot_index(0x1234_5678, 1 << 24), 0x1234_5678 >> 8); + assert_eq!(h.word(0x8000_0000), h.at(1 << 22)); + } + + #[test] + fn first_cache_line_matches_pack() { + // vectors.json cache_head for day 2026-10-03: segment 0, line 0, with prev = 0. + let key = day_key("2026-10-03"); + let mut words = vec![0u32; CACHE_LINES_PER_SEGMENT * 16]; + Cache::fill_segment(&mut words, 0, &key); + assert_eq!( + &words[..16], + &[ + 0x355a86d2, 0x7957db1c, 0xd21772af, 0x6fc1e09b, 0xd55ce61d, 0x6e6a278b, 0xd3f543ce, 0x223d8e82, + 0x143ab337, 0x2e9f05bd, 0x2eb389bf, 0x0c6e449e, 0x5cfa4222, 0xba6560fe, 0x8e3e1aa4, 0xdbcc1d53 + ] + ); + } + + /// Option C: the schedule table of `docs/plans/mixer-x4.md` (day -> doublings, cache words, dataset words at a + /// 2^28 genesis). The doublings fall at years 4 and 12 exactly, never a day early. + #[test] + fn growth_schedule_table() { + let table: [(u64, u32, u32, u32); 12] = [ + (0, 0, 26, 28), + (1, 0, 26, 28), + (365, 0, 26, 28), + (1_459, 0, 26, 28), + (1_460, 1, 27, 29), + (2_920, 1, 27, 29), + (4_379, 1, 27, 29), + (4_380, 2, 28, 30), + (10_219, 2, 28, 30), + (10_220, 3, 29, 31), + (21_900, 4, 30, 32), + (100_000, 6, 32, 32), + ]; + for (d, k, c, s) in table { + assert_eq!(growth_doublings(d), k, "day {d}"); + assert_eq!(cache_log2_words(d), c, "day {d}"); + assert_eq!(dataset_log2_words(28, d), s, "day {d}"); + } + // the designed 2 GiB genesis: 2^29 words, 2^30 at year 4, 2^31 at year 12 + assert_eq!(dataset_log2_words(29, 0), 29); + assert_eq!(dataset_log2_words(29, 1_460), 30); + assert_eq!(dataset_log2_words(29, 4_380), 31); + // the linear schedule itself: 2 GiB x (1 + d / 1460) crosses 4 GiB at day 1,460 and 8 GiB at day 4,380 + for d in [1_459u64, 1_460, 4_379, 4_380] { + let bytes = 2u64 * (1 << 30) + (1u64 << 29) * d / 365; + let k = (bytes / (2u64 << 30)).ilog2(); + assert_eq!(growth_doublings(d), k, "day {d}: linear {bytes} bytes"); + } + assert_eq!(days_since_genesis(20_730, 20_729), 1); + assert_eq!(days_since_genesis(20_729, 20_729), 0); + assert_eq!(days_since_genesis(20_000, 20_729), 0); + let v2 = Shape::for_class_day(&LoadClass::V2, 100_000); + assert_eq!(v2, Shape::V2); + let v3 = Shape::for_class_day(&LoadClass::MX4, 0); + assert_eq!(v3, Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0, state: false }); + assert_eq!(Shape::for_class_day(&LoadClass::MX4, 1_460).cache_log2_words, 27); + assert_eq!(v3.mixers_per_item(), 36); + assert_eq!(Shape::V2.mixers_per_item(), 9); + assert_eq!(Shape::V2.cache_segments(), CACHE_SEGMENTS); + assert_eq!(Shape::V2.cache_line_mask(), CACHE_LINE_MASK); + assert_eq!(Shape::V2.log2_segments(), 16); + } + + /// The multiplied mixer, restated by hand on a small cache: `m` applications with keys `round_key(r m + j)` + /// before every read, the same 8 reads; `m = 1` is `derive_item` of version 2 word for word; a larger cache's + /// first segments equal the smaller cache's. + #[test] + fn mixer_mult_by_hand() { + let key = day_key("2026-10-03"); + let small = Cache::fill_log2(key, 16); + let big = Cache::fill_log2(key, 18); + assert_eq!(&big.words()[..small.words().len()], small.words()); + assert_eq!(small.segments(), 64); + assert_eq!(small.line_mask(), 4095); + for m in [1u32, 2, 4] { + let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 16, derive_len: 0, state: false }); + for t in [0u32, 1, 12_345, u32::MAX] { + let got = derive_item(t, &mp, &small); + let mut s = [0u32; 16]; + s[..8].copy_from_slice(&key); + for i in 0..8 { + s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); + } + for r in 0..8usize { + for j in 0..m as usize { + mixer(&mut s, round_key(r * m as usize + j), &mp); + } + let line = small.line(s[0]); + for i in 0..16 { + s[i] ^= line[i]; + } + } + for j in 0..m as usize { + mixer(&mut s, round_key(8 * m as usize + j), &mp); + } + assert_eq!(got, s, "m {m} t {t}"); + } + } + let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0, state: false }); + let v3 = MixParams::with_shape(key, Shape { mixer_mult: 4, cache_log2_words: 16, derive_len: 0, state: false }); + assert_ne!(derive_item(0, &v2, &small), derive_item(0, &v3, &small)); + assert_eq!(round_key_mult(0, 0, 1), round_key(0)); + assert_eq!(round_key_mult(8, 0, 1), round_key(8)); + assert_eq!(round_key_mult(2, 3, 4), round_key(11)); + assert_eq!(round_key_mult(8, 3, 4), round_key(35)); + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/packcheck.rs b/tools/attack/adv-accept-v5/igneum-pow/src/packcheck.rs new file mode 100644 index 000000000..a7cbea7f5 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/packcheck.rs @@ -0,0 +1,464 @@ +//! A program pack on disk, read the way the one-click workers read it (`proto-cuda/nvrtc/packfile.h`, `pf_load`), +//! and checked against the seeds the node is on. +//! +//! The rule (5 October 2026, the epoch 34 incident on both PCs): a pack's `IGNEUM_SEEDW_INIT` is the seed words of +//! the program's ATTEMPT, `attempt_words(epoch_seed, IGNEUM_PROGRAM_ATTEMPT)`, not the words of the bare seed. The +//! generator retries a rejected candidate with `seed || k_le32` (spec 01 section 1.4.6), so from attempt 1 on the +//! bare-seed words and the pack's words differ. The workers derived the expected words from the bare seed, refused +//! every pack of a retried program ("the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT") and the miner +//! and the app restarted them forever. Epoch 34 (seed `009858237e11...`) was the first live epoch whose program is +//! a later attempt. This module is the one place that rule is written in Rust; the miner checks every pack it +//! writes with it before a worker sees the pack, and the tests pin the attempt vectors the C side also pins. + +use crate::generator::{attempt_words, ProgramClass}; +use crate::seed::seed_words_from_bytes; +use std::fmt; +use std::path::Path; + +/// What a pack says about itself. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PackIdentity { + pub epoch_hex: String, + pub day_hex: String, + pub attempt: u32, + pub seedw: [u32; 8], + pub keyw: [u32; 8], + /// `IGNEUM_GENERATOR` (2, 3 or 4; a pack without the line is generator 1, which no worker runs). + pub generator: u32, + /// The program class the generator version names (Counter ASIC 2.0). + pub class: ProgramClass, + /// `IGNEUM_ERA_SEED_HEX` when the pack carries one (class v3 chain packs). + pub era_hex: Option, + /// `IGNEUM_STATE_ROOT_HEX` of a class v5 pack (the window's state root the leaves derive from). + pub state_root_hex: Option, +} + +/// Why a pack is not the one a worker should mine with. `Display` is the plain-words line the logs carry. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum PackFault { + /// program.h or seeds.txt is missing or does not parse. + Unreadable(String), + /// A well-formed pack for other seeds than the node's: the pack is stale (or the node moved on). + OutOfDate { pack_epoch: String, pack_day: String, want_epoch: String, want_day: String }, + /// The files of one pack contradict each other (seeds.txt against program.h, or the init words against the + /// seeds and the attempt): a half-written or hand-edited pack, or a worker and an exporter on different rules. + Disagree(String), + /// The pack is of another program class than the one the chain is on (spec 01 section 1.4.5: an implementation + /// refuses a pack whose generator version is not its own), or its era seed is not the era the job names. + WrongClass(String), +} + +impl fmt::Display for PackFault { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + PackFault::Unreadable(w) => write!(f, "program pack unreadable: {w}"), + PackFault::OutOfDate { pack_epoch, pack_day, want_epoch, want_day } => write!( + f, + "program pack out of date: the pack is for epoch {} day {}, the node is on epoch {} day {}", + short(pack_epoch), + day_label(pack_day), + short(want_epoch), + day_label(want_day) + ), + PackFault::Disagree(w) => write!(f, "program pack and its seeds disagree: {w}"), + PackFault::WrongClass(w) => write!(f, "program pack of the wrong class: {w}"), + } + } +} + +impl std::error::Error for PackFault {} + +fn short(hex: &str) -> &str { + if hex.len() >= 16 { + &hex[..16] + } else { + hex + } +} + +/// The day bytes are `igneum-day/` followed by the little-endian day index (`bind::day_bytes`); print the index +/// when the hex has that shape, else the hex. +fn day_label(hex: &str) -> String { + const PREFIX: &str = "69676e65756d2d6461792f"; // "igneum-day/" + if let Some(rest) = hex.strip_prefix(PREFIX) { + if let Some(bytes) = unhex(rest) { + let mut v = 0u64; + for (i, b) in bytes.iter().enumerate().take(8) { + v |= (*b as u64) << (8 * i); + } + return v.to_string(); + } + } + hex.to_string() +} + +pub fn hex(bytes: &[u8]) -> String { + bytes.iter().map(|b| format!("{b:02x}")).collect() +} + +fn unhex(s: &str) -> Option> { + if s.len() % 2 != 0 { + return None; + } + (0..s.len()).step_by(2).map(|i| u8::from_str_radix(&s[i..i + 2], 16).ok()).collect() +} + +/// `#define NAME ` in a header; the value as text, trimmed, with a trailing `//` comment removed. +fn define(text: &str, name: &str) -> Option { + for line in text.lines() { + let t = line.trim_start(); + let Some(rest) = t.strip_prefix("#define ") else { continue }; + let rest = rest.trim_start(); + let Some(after) = rest.strip_prefix(name) else { continue }; + if !after.starts_with(|c: char| c.is_whitespace()) { + continue; + } + let v = after.trim(); + let v = v.split("//").next().unwrap_or("").trim(); + return Some(v.to_string()); + } + None +} + +fn define_str(text: &str, name: &str) -> Option { + let v = define(text, name)?; + let v = v.strip_prefix('"')?.strip_suffix('"')?; + Some(v.to_string()) +} + +fn define_u32(text: &str, name: &str) -> Option { + let v = define(text, name)?; + let v = v.trim_end_matches('u'); + if let Some(h) = v.strip_prefix("0x") { + u32::from_str_radix(h, 16).ok() + } else { + v.parse().ok() + } +} + +fn define_words(text: &str, name: &str) -> Option<[u32; 8]> { + let v = define(text, name)?; + let inner = v.trim().strip_prefix('{')?.strip_suffix('}')?; + let mut out = [0u32; 8]; + let mut n = 0; + for part in inner.split(',') { + let p = part.trim().trim_end_matches('u'); + if p.is_empty() { + continue; + } + if n >= 8 { + return None; + } + out[n] = if let Some(h) = p.strip_prefix("0x") { u32::from_str_radix(h, 16).ok()? } else { p.parse().ok()? }; + n += 1; + } + (n == 8).then_some(out) +} + +/// One `key value` line of seeds.txt. +fn seeds_line(text: &str, key: &str) -> Option { + text.lines().find_map(|l| l.strip_prefix(key).and_then(|r| r.strip_prefix(' ')).map(|v| v.trim().to_string())) +} + +/// Checks the texts of a pack (program.h, and seeds.txt when it exists) against the seeds a worker will be asked +/// to mine with. Pure: the miner and the tests call it with file contents. +pub fn verify_pack_texts(program_h: &str, seeds_txt: Option<&str>, want_epoch: &[u8], want_day: &[u8]) -> Result { + verify_pack_texts_chain(program_h, seeds_txt, want_epoch, want_day, None, None) +} + +/// [`verify_pack_texts`] that also demands a program class and, for class v3 and v4, the era seed the chain is on +/// (Counter ASIC 2.0, 5 October 2026; class v4 Counter ASIC 3.0, 6 October 2026). `want_class` `None` accepts any +/// class; `want_era` `None` skips the era. A pack whose `IGNEUM_GENERATOR` is not 2, 3 or 4 is refused whatever is wanted. +pub fn verify_pack_texts_chain( + program_h: &str, + seeds_txt: Option<&str>, + want_epoch: &[u8], + want_day: &[u8], + want_class: Option, + want_era: Option<&[u8]>, +) -> Result { + let generator = define_u32(program_h, "IGNEUM_GENERATOR").unwrap_or(1); + let Some(class) = ProgramClass::from_generator(generator) else { + return Err(PackFault::WrongClass(format!("IGNEUM_GENERATOR {generator} is not a generator version this software runs (2, 3 or 4)"))); + }; + // IGNEUM_PROGRAM_CLASS, when present, must name the class the generator version names + if let Some(named) = define_str(program_h, "IGNEUM_PROGRAM_CLASS") { + if ProgramClass::parse(&named) != Some(class) { + return Err(PackFault::Disagree(format!("IGNEUM_PROGRAM_CLASS {named:?} does not match IGNEUM_GENERATOR {generator}"))); + } + } + let era_hex = define_str(program_h, "IGNEUM_ERA_SEED_HEX").map(|h| h.to_ascii_lowercase()); + // Counter ASIC 3.0 (6 October 2026): the shadow block is the mark of class v4, so a generator 4 pack carries + // IGNEUM_SHADOW_INSTRS and a generator 3 pack does not; a pack exported as class v4 but stamped generator 3 (the + // seven gate packs of 6 October, which carried the v3 control's program id because `program_id(3, ..)` is + // class-independent) is refused here instead of mining as class v3 under the wrong id, and a generator 4 pack + // without the block is no v4 pack. A generator 2 pack with a shadow (the measurement ladder of + // proto-cuda/packs-ca3-shadow) carries a class-bearing id and stays loadable. + let shadow = define_u32(program_h, "IGNEUM_SHADOW_INSTRS").unwrap_or(0); + // Class v5 (docs/design/class-v5-stored-state.md): the state lines are the mark of class v5, so a generator 5 pack + // carries IGNEUM_STATE_ROOT_HEX (and the shadow block of v4) and no other generator does. + let state_root_hex = define_str(program_h, "IGNEUM_STATE_ROOT_HEX"); + match (class, state_root_hex.is_some()) { + (ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_STATE_ROOT_HEX: not a class v5 pack".into())), + (ProgramClass::V5, true) => {} + (_, true) => return Err(PackFault::Disagree(format!("IGNEUM_GENERATOR {generator} with class v5 state lines: a program over state leaves is generator 5 (export the pack as class v5)"))), + _ => {} + } + match (class, shadow > 0) { + (ProgramClass::V4, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 4 (class v4) without IGNEUM_SHADOW_INSTRS: not a class v4 pack".into())), + (ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_SHADOW_INSTRS: not a class v5 pack".into())), + // the class v4 stream sub-version (AP-F8-1 amendment, 7 October 2026): a generator 4 pack from before the + // load-source rule carries no IGNEUM_PROGRAM_SUBVERSION and its program id is another stream's; refused + (ProgramClass::V4, true) if define_u32(program_h, "IGNEUM_PROGRAM_SUBVERSION") != Some(u32::from(crate::generator::PROGRAM_SUBVERSION_V4)) => { + return Err(PackFault::Disagree(format!( + "IGNEUM_GENERATOR 4 (class v4) with IGNEUM_PROGRAM_SUBVERSION {}: this software runs sub-version {} (a class v4 pack from before the load-source rule, or after another amendment)", + define_u32(program_h, "IGNEUM_PROGRAM_SUBVERSION").map(|v| v.to_string()).unwrap_or_else(|| "absent".into()), + crate::generator::PROGRAM_SUBVERSION_V4 + ))) + } + (ProgramClass::V3, true) => { + return Err(PackFault::Disagree(format!("IGNEUM_GENERATOR 3 (class v3) with a shadow block (IGNEUM_SHADOW_INSTRS {shadow}): a class v4 program is generator 4 (export the pack as class v4)"))) + } + _ => {} + } + if let Some(want) = want_class { + if want != class { + return Err(PackFault::WrongClass(format!("the pack is program class {} (generator {generator}), the chain is on class {}", class.name(), want.name()))); + } + } + if let (Some(want), true) = (want_era, class.has_era()) { + let want_hex = hex(want); + match &era_hex { + Some(h) if *h == want_hex => {} + Some(h) => return Err(PackFault::WrongClass(format!("the pack's era seed {} is not the era seed {} the job names", short(h), short(&want_hex)))), + None => return Err(PackFault::WrongClass(format!("a class {} pack without IGNEUM_ERA_SEED_HEX; the job names an era seed", class.name()))), + } + } + let seedw = define_words(program_h, "IGNEUM_SEEDW_INIT").ok_or_else(|| PackFault::Unreadable("program.h has no IGNEUM_SEEDW_INIT with 8 words".into()))?; + let keyw = define_words(program_h, "IGNEUM_KEY_INIT").ok_or_else(|| PackFault::Unreadable("program.h has no IGNEUM_KEY_INIT with 8 words".into()))?; + let attempt = define_u32(program_h, "IGNEUM_PROGRAM_ATTEMPT").unwrap_or(0); + let mut epoch_hex = define_str(program_h, "IGNEUM_SEED_BYTES_HEX").unwrap_or_default(); + let mut day_hex = define_str(program_h, "IGNEUM_DAY_BYTES_HEX").unwrap_or_default(); + if let Some(s) = seeds_txt { + let e = seeds_line(s, "epoch_seed_hex").ok_or_else(|| PackFault::Unreadable("seeds.txt has no epoch_seed_hex line".into()))?; + let d = seeds_line(s, "day_seed_hex").ok_or_else(|| PackFault::Unreadable("seeds.txt has no day_seed_hex line".into()))?; + if !epoch_hex.is_empty() && !epoch_hex.eq_ignore_ascii_case(&e) { + return Err(PackFault::Disagree(format!("seeds.txt names epoch {} but program.h was generated for epoch {} (a pack half rewritten?)", short(&e), short(&epoch_hex)))); + } + if !day_hex.is_empty() && !day_hex.eq_ignore_ascii_case(&d) { + return Err(PackFault::Disagree(format!("seeds.txt names day {} but program.h was generated for day {}", day_label(&d), day_label(&day_hex)))); + } + epoch_hex = e.to_ascii_lowercase(); + day_hex = d.to_ascii_lowercase(); + } + if epoch_hex.is_empty() || day_hex.is_empty() { + return Err(PackFault::Unreadable("no seeds: neither seeds.txt nor IGNEUM_SEED_BYTES_HEX / IGNEUM_DAY_BYTES_HEX in program.h".into())); + } + let epoch_bytes = unhex(&epoch_hex).filter(|b| b.len() == 32).ok_or_else(|| PackFault::Unreadable("the epoch seed is not 32 bytes of hex".into()))?; + let day_bytes = unhex(&day_hex).ok_or_else(|| PackFault::Unreadable("the day seed hex is malformed".into()))?; + // The pack's own consistency first: a pack that contradicts itself is never "out of date", it is broken + let want_w = attempt_words(&epoch_bytes, attempt); + if want_w != seedw { + return Err(PackFault::Disagree(format!( + "IGNEUM_SEEDW_INIT is not attempt {attempt} of the epoch seed {} (the words of attempt {attempt} are {:08x} {:08x} ..., the pack has {:08x} {:08x} ...)", + short(&epoch_hex), + want_w[0], + want_w[1], + seedw[0], + seedw[1] + ))); + } + let want_k = seed_words_from_bytes(&day_bytes); + if want_k != keyw { + return Err(PackFault::Disagree(format!("IGNEUM_KEY_INIT is not the key of the day seed {} ", day_label(&day_hex)))); + } + let want_epoch_hex = hex(want_epoch); + let want_day_hex = hex(want_day); + if epoch_hex != want_epoch_hex || day_hex != want_day_hex { + return Err(PackFault::OutOfDate { pack_epoch: epoch_hex, pack_day: day_hex, want_epoch: want_epoch_hex, want_day: want_day_hex }); + } + Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw, generator, class, era_hex, state_root_hex }) +} + +/// [`verify_pack_texts`] over a pack directory. +pub fn verify_pack_dir(dir: &Path, want_epoch: &[u8], want_day: &[u8]) -> Result { + verify_pack_dir_chain(dir, want_epoch, want_day, None, None) +} + +/// [`verify_pack_texts_chain`] over a pack directory. +pub fn verify_pack_dir_chain(dir: &Path, want_epoch: &[u8], want_day: &[u8], want_class: Option, want_era: Option<&[u8]>) -> Result { + let program_h = std::fs::read_to_string(dir.join("program.h")).map_err(|e| PackFault::Unreadable(format!("cannot read {}/program.h: {e}", dir.display())))?; + let seeds = std::fs::read_to_string(dir.join("seeds.txt")).ok(); + verify_pack_texts_chain(&program_h, seeds.as_deref(), want_epoch, want_day, want_class, want_era) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::emit::program_header; + use crate::generator::generate_from_seed_bytes; + use crate::verify::Epoch; + + // The two live devnet epochs of 5 October 2026 (epoch 33 mined, epoch 34 refused by the one-click workers) + const EPOCH_33: &str = "bed7ab62cbece66cf791485336d81d90fa1452ffed28ecd8a7416960ef64164c"; + const EPOCH_34: &str = "009858237e118f69abc8d096e9b1af21c24539eaecdfd1b896588825660a69ec"; + // "igneum-day/" || le64(20731) + const DAY_20731: &str = "69676e65756d2d6461792ffb50000000000000"; + + fn bytes(h: &str) -> Vec { + unhex(h).unwrap() + } + + /// The attempt vectors the C side pins too (proto-cuda/nvrtc/emu/packfile-test.c): a change to either + /// derivation fails on one side first. + #[test] + fn attempt_words_vectors_shared_with_the_workers() { + let e = bytes(EPOCH_34); + assert_eq!(attempt_words(&e, 0), [0x06af2a61, 0x4d67274e, 0x4ebda738, 0xad1dea73, 0x6233cd8c, 0x50371601, 0x39d0b873, 0x6af024a2]); + assert_eq!(attempt_words(&e, 1), [0x0dcff56b, 0x6b1beb0d, 0x234dc70c, 0xe4016fa9, 0x72397152, 0xb558aa79, 0x3ffb3299, 0x72b9962e]); + // epoch 33's bare words, as the CUDA worker printed them on 5 October ("seed words be5983a6 f750dab7 ...") + assert_eq!(attempt_words(&bytes(EPOCH_33), 0)[..2], [0xbe5983a6, 0xf750dab7]); + } + + /// The incident: epoch 34's program is a later attempt, epoch 33's is the bare seed. A worker that derives the + /// words from the bare seed accepts 33 and refuses 34. + #[test] + fn epoch_34_program_is_a_later_attempt() { + let p34 = generate_from_seed_bytes("epoch 34", &bytes(EPOCH_34)); + assert!(p34.attempt >= 1, "epoch 34 must be a retried program for the incident to reproduce; attempt {}", p34.attempt); + assert_eq!(p34.seed, attempt_words(&bytes(EPOCH_34), p34.attempt)); + assert_ne!(p34.seed, attempt_words(&bytes(EPOCH_34), 0)); + let p33 = generate_from_seed_bytes("epoch 33", &bytes(EPOCH_33)); + assert_eq!(p33.attempt, 0); + } + + fn pack_texts(epoch_hex: &str, day_hex: &str) -> (String, String, u32) { + let (e, d) = (bytes(epoch_hex), bytes(day_hex)); + let epoch = Epoch::from_seed_bytes(&e, &d, "test"); + let h = program_header(&epoch.program, "test day", &epoch.dataset); + let s = format!("epoch_seed_hex {epoch_hex}\nday_seed_hex {day_hex}\nday_index 20731\n"); + (h, s, epoch.program.attempt) + } + + /// Known-good: the pack of a retried program verifies against its own seeds, with its attempt. + #[test] + fn known_good_pack_of_a_later_attempt_verifies() { + let (h, s, attempt) = pack_texts(EPOCH_34, DAY_20731); + assert!(attempt >= 1); + let id = verify_pack_texts(&h, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).expect("the pack verifies"); + assert_eq!(id.attempt, attempt); + assert_eq!(id.epoch_hex, EPOCH_34); + assert_eq!(id.seedw, attempt_words(&bytes(EPOCH_34), attempt)); + // without seeds.txt program.h's own bytes carry the pack + assert!(verify_pack_texts(&h, None, &bytes(EPOCH_34), &bytes(DAY_20731)).is_ok()); + } + + /// Known-mismatched: a well-formed pack for the previous epoch is "out of date" against the new one, in plain + /// words with both epochs named. + #[test] + fn known_mismatched_pack_is_out_of_date() { + let (h, s, _) = pack_texts(EPOCH_33, DAY_20731); + let err = verify_pack_texts(&h, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err(); + assert!(matches!(err, PackFault::OutOfDate { .. }), "{err}"); + assert_eq!(err.to_string(), "program pack out of date: the pack is for epoch bed7ab62cbece66c day 20731, the node is on epoch 009858237e118f69 day 20731"); + } + + /// A pack that contradicts itself is "disagree", never "out of date": seeds.txt of one epoch with program.h of + /// another (a half rewritten directory), or init words that are not the attempt's words (a worker on the old + /// rule would have produced this verdict for every retried program). + #[test] + fn inconsistent_pack_disagrees() { + let (h33, _, _) = pack_texts(EPOCH_33, DAY_20731); + let s34 = format!("epoch_seed_hex {EPOCH_34}\nday_seed_hex {DAY_20731}\n"); + let err = verify_pack_texts(&h33, Some(&s34), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err(); + assert!(matches!(err, PackFault::Disagree(_)), "{err}"); + assert!(err.to_string().starts_with("program pack and its seeds disagree: seeds.txt names epoch 009858237e118f69"), "{err}"); + + let (h34, s, _) = pack_texts(EPOCH_34, DAY_20731); + let bare = attempt_words(&bytes(EPOCH_34), 0); + let edited = h34.lines().map(|l| if l.starts_with("#define IGNEUM_SEEDW_INIT") { format!("#define IGNEUM_SEEDW_INIT {{ {} }}", bare.iter().map(|w| format!("0x{w:08x}")).collect::>().join(", ")) } else { l.to_string() }).collect::>().join("\n"); + let err = verify_pack_texts(&edited, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err(); + assert!(err.to_string().contains("IGNEUM_SEEDW_INIT is not attempt"), "{err}"); + // and a pack with no attempt line at all is read as attempt 0 (the packs before generator version 2) + let no_attempt = h34.lines().filter(|l| !l.starts_with("#define IGNEUM_PROGRAM_ATTEMPT")).collect::>().join("\n"); + assert!(verify_pack_texts(&no_attempt, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).is_err()); + } + + #[test] + fn day_label_reads_the_index() { + assert_eq!(day_label(DAY_20731), "20731"); + assert_eq!(day_label("abcd"), "abcd"); + } + + /// Counter ASIC 2.0: a class v3 chain pack carries generator 3, the class line and the era seed; it is refused + /// when the chain wants class v2, when the era differs, and a v2 pack is refused when the chain wants v3; a + /// generator this software does not run is refused whatever is wanted. + #[test] + fn program_class_and_era_are_checked() { + let e = bytes(EPOCH_34); + let d = bytes(DAY_20731); + let era = [0x5au8; 32]; + let v3 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V3, "class test"); + let h3 = program_header(&v3.program, "test day", &v3.dataset); + assert!(h3.contains("#define IGNEUM_GENERATOR 3\n")); + assert!(h3.contains("#define IGNEUM_PROGRAM_CLASS \"v3\"\n")); + assert!(h3.contains(&format!("#define IGNEUM_ERA_SEED_HEX \"{}\"\n", hex(&era)))); + let id = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V3), Some(&era)).unwrap(); + assert_eq!((id.generator, id.class, id.era_hex.as_deref()), (3, ProgramClass::V3, Some(hex(&era).as_str()))); + assert_eq!(id.attempt, v3.program.attempt); + assert!(verify_pack_texts(&h3, None, &e, &d).is_ok(), "no class wanted: either class passes"); + let err = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V2), None).unwrap_err(); + assert!(matches!(err, PackFault::WrongClass(_)), "{err}"); + assert!(err.to_string().contains("program pack of the wrong class"), "{err}"); + let err = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V3), Some(&[1u8; 32])).unwrap_err(); + assert!(err.to_string().contains("era seed"), "{err}"); + // the v2 pack of the same seeds: generator 2, no class line, no era line, refused when v3 is wanted + let v2 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V2, "class test"); + let h2 = program_header(&v2.program, "test day", &v2.dataset); + assert!(h2.contains("#define IGNEUM_GENERATOR 2\n")); + assert!(!h2.contains("IGNEUM_PROGRAM_CLASS") && !h2.contains("IGNEUM_ERA_SEED_HEX")); + let plain = Epoch::from_seed_bytes(&e, &d, "class test"); + assert_eq!(program_header(&plain.program, "test day", &plain.dataset), h2, "class v2 from the chain is the v2 export byte for byte"); + let id = verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V2), Some(&era)).unwrap(); + assert_eq!((id.generator, id.class, id.era_hex), (2, ProgramClass::V2, None)); + assert!(matches!(verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V3), None), Err(PackFault::WrongClass(_)))); + // a generator nobody runs + let h9 = h2.replace("#define IGNEUM_GENERATOR 2\n", "#define IGNEUM_GENERATOR 9\n"); + assert!(matches!(verify_pack_texts(&h9, None, &e, &d), Err(PackFault::WrongClass(_)))); + // a class line that contradicts the generator + let bad = h3.replace("#define IGNEUM_PROGRAM_CLASS \"v3\"\n", "#define IGNEUM_PROGRAM_CLASS \"v2\"\n"); + assert!(matches!(verify_pack_texts(&bad, None, &e, &d), Err(PackFault::Disagree(_)))); + // Counter ASIC 3.0: a class v4 chain pack carries generator 4, the class line and the era seed; it is refused + // when v3 (or v2) is wanted, and a v3 pack is refused when v4 is wanted; the era rule applies to v4 as to v3 + let v4 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V4, "class test"); + let h4 = program_header(&v4.program, "test day", &v4.dataset); + assert!(h4.contains("#define IGNEUM_GENERATOR 4\n")); + assert!(h4.contains("#define IGNEUM_PROGRAM_CLASS \"v4\"\n")); + assert!(h4.contains(&format!("#define IGNEUM_ERA_SEED_HEX \"{}\"\n", hex(&era)))); + assert!(h4.contains("#define IGNEUM_SHADOW_INSTRS 256\n") && h4.contains("#define IGNEUM_SHADOW_REPS 27\n"), "{h4}"); + let id = verify_pack_texts_chain(&h4, None, &e, &d, Some(ProgramClass::V4), Some(&era)).unwrap(); + assert_eq!((id.generator, id.class, id.era_hex.as_deref()), (4, ProgramClass::V4, Some(hex(&era).as_str()))); + assert_eq!(id.attempt, v3.program.attempt, "the v4 attempt is the v3 attempt of the same seed"); + assert!(verify_pack_texts(&h4, None, &e, &d).is_ok(), "no class wanted: any class passes"); + assert!(matches!(verify_pack_texts_chain(&h4, None, &e, &d, Some(ProgramClass::V3), None), Err(PackFault::WrongClass(_)))); + assert!(matches!(verify_pack_texts_chain(&h4, None, &e, &d, Some(ProgramClass::V2), None), Err(PackFault::WrongClass(_)))); + assert!(matches!(verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V4), None), Err(PackFault::WrongClass(_)))); + assert!(matches!(verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V4), None), Err(PackFault::WrongClass(_)))); + let err = verify_pack_texts_chain(&h4, None, &e, &d, Some(ProgramClass::V4), Some(&[1u8; 32])).unwrap_err(); + assert!(err.to_string().contains("era seed"), "{err}"); + let no_era = h4.replace(&format!("#define IGNEUM_ERA_SEED_HEX \"{}\"\n", hex(&era)), ""); + let err = verify_pack_texts_chain(&no_era, None, &e, &d, Some(ProgramClass::V4), Some(&era)).unwrap_err(); + assert!(err.to_string().contains("class v4 pack without IGNEUM_ERA_SEED_HEX"), "{err}"); + // the v3 pack of the same seeds is unchanged by the v4 class existing + assert_eq!(program_header(&v3.program, "test day", &v3.dataset), h3); + // the 6 October trap: a v4 program stamped generator 3 (the CLI's old --era path) is refused for its shadow + // block, whatever class is wanted; and a generator 4 header without the block is refused too + let stamped3 = h4.replace("#define IGNEUM_GENERATOR 4\n", "#define IGNEUM_GENERATOR 3\n").replace("#define IGNEUM_PROGRAM_CLASS \"v4\"\n", "#define IGNEUM_PROGRAM_CLASS \"v3\"\n"); + let err = verify_pack_texts(&stamped3, None, &e, &d).unwrap_err(); + assert!(matches!(err, PackFault::Disagree(_)) && err.to_string().contains("shadow block"), "{err}"); + assert!(verify_pack_texts_chain(&stamped3, None, &e, &d, Some(ProgramClass::V3), Some(&era)).is_err()); + let no_shadow = h4.replace("#define IGNEUM_SHADOW_INSTRS 256\n", ""); + let err = verify_pack_texts(&no_shadow, None, &e, &d).unwrap_err(); + assert!(matches!(err, PackFault::Disagree(_)) && err.to_string().contains("without IGNEUM_SHADOW_INSTRS"), "{err}"); + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/seed.rs b/tools/attack/adv-accept-v5/igneum-pow/src/seed.rs new file mode 100644 index 000000000..68d40911a --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/seed.rs @@ -0,0 +1,114 @@ +//! Seed derivation and the SplitMix64 stream, exactly as `proto-metal/main.swift` does them. +//! +//! Today a seed is a string ("igneum-genesis", "day/2026-10-03"). On the chain the epoch seed will be the +//! output of a class-group VDF over a certified checkpoint hash. [`seed_words_from_bytes`] is the function +//! boundary for that: whatever bytes the chain settles on go through the same FNV-1a construction, so the +//! generator and the day key never need to know where the bytes came from. + +/// FNV-1a 64 over `bytes` with the standard basis. Used for the cache fingerprint in the packs. +pub fn fnv1a64(bytes: &[u8]) -> u64 { + fnv1a64_with_basis(0xcbf29ce484222325, bytes) +} + +#[inline] +fn fnv1a64_with_basis(basis: u64, bytes: &[u8]) -> u64 { + let mut h = basis; + for &b in bytes { + h ^= b as u64; + h = h.wrapping_mul(0x100000001b3); + } + h +} + +/// FNV-1a 64 over 32-bit words in little-endian byte order (the cache is hashed as raw memory). +pub fn fnv1a64_words(words: &[u32]) -> u64 { + let mut h: u64 = 0xcbf29ce484222325; + for &w in words { + for b in w.to_le_bytes() { + h ^= b as u64; + h = h.wrapping_mul(0x100000001b3); + } + } + h +} + +/// The 32-byte seed (8 x u32) from arbitrary bytes: FNV-1a 64 with four salts, each finalised with the +/// murmur-style mix `h ^= h >> 33; h *= 0xff51afd7ed558ccd; h ^= h >> 33`; low word then high word. +pub fn seed_words_from_bytes(bytes: &[u8]) -> [u32; 8] { + let mut words = [0u32; 8]; + for salt in 0..4u64 { + let basis = 0xcbf29ce484222325u64 ^ salt.wrapping_mul(0x9E3779B97F4A7C15); + let mut h = fnv1a64_with_basis(basis, bytes); + h ^= h >> 33; + h = h.wrapping_mul(0xff51afd7ed558ccd); + h ^= h >> 33; + words[2 * salt as usize] = h as u32; + words[2 * salt as usize + 1] = (h >> 32) as u32; + } + words +} + +/// The 32-byte seed from a string (its UTF-8 bytes). `seedWords` in the Swift. +pub fn seed_words(s: &str) -> [u32; 8] { + seed_words_from_bytes(s.as_bytes()) +} + +/// The day key: the 8 words of `seed_words("day/" + day)`. `K` in MEMHARD.md; `d0, d1` are `K[0], K[1]`. +pub fn day_key(day: &str) -> [u32; 8] { + seed_words(&format!("day/{day}")) +} + +/// SplitMix64, the one deterministic stream every draw in the prototype comes from. +#[derive(Clone, Copy, Debug)] +pub struct SplitMix64 { + pub s: u64, +} + +impl SplitMix64 { + pub fn new(s: u64) -> Self { + Self { s } + } + #[inline] + pub fn next(&mut self) -> u64 { + self.s = self.s.wrapping_add(0x9E3779B97F4A7C15); + let mut z = self.s; + z = (z ^ (z >> 30)).wrapping_mul(0xBF58476D1CE4E5B9); + z = (z ^ (z >> 27)).wrapping_mul(0x94D049BB133111EB); + z ^ (z >> 31) + } + /// `next() % n` as the Swift `below` does it (modulo, not rejection sampling). + #[inline] + pub fn below(&mut self, n: u64) -> u64 { + self.next() % n + } +} + +/// The generator's stream for a seed: `(w0 | w1 << 32) ^ ((w2 | w3 << 32) * 0x9E3779B97F4A7C15)`. +pub fn program_rng(seed: &[u32; 8]) -> SplitMix64 { + let lo = seed[0] as u64 | ((seed[1] as u64) << 32); + let hi = seed[2] as u64 | ((seed[3] as u64) << 32); + SplitMix64::new(lo ^ hi.wrapping_mul(0x9E3779B97F4A7C15)) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn genesis_seed_words() { + // From proto-cuda/packs/igneum-genesis/program.json. + assert_eq!( + seed_words("igneum-genesis"), + [0x67a9a7be, 0x1a155b25, 0xfddfb732, 0x4b5af2e8, 0xc55caf33, 0xa27c13b7, 0x06628a48, 0x03852469] + ); + } + + #[test] + fn day_key_2026_10_03() { + // MEMHARD.md section 1.1. + assert_eq!( + day_key("2026-10-03"), + [0x3067619f, 0x3c269176, 0x84a03b03, 0xf8c63294, 0xff977c5b, 0xe60def3e, 0x63630141, 0xb8fbcb58] + ); + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/state.rs b/tools/attack/adv-accept-v5/igneum-pow/src/state.rs new file mode 100644 index 000000000..4298b7fde --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/state.rs @@ -0,0 +1,287 @@ +//! Class v5, proof of stored state and of following (`docs/design/class-v5-stored-state.md`, 7 October 2026): the +//! leaves the item derivation XORs in (section 2 of the page), built from the canonical state stream of the +//! window's reference block. +//! +//! `D[i] = Blake2b-512("igneum-sd1/" || root || i_le32 || record_i)` for the `n` records of the stream, and item +//! `t` takes `leaf(t) = D[t mod n]`: every item is keyed by the state, so a hasher without it is wrong on every +//! item (the known-failed case, the first test). When the stream has more records than the dataset has items, the +//! records are ordered by `Blake2b-256("igneum-sd1-sample/" || root || record)` and the first `items` are taken, a +//! sample nobody can choose without the whole state and the root. +//! +//! The stream file (`StateStream`): the plain format every side reads without a serialisation library, `IGSD1\0`, +//! the chain block number (le64) and hash (32), the state root (32), the record count (le32), then each record as +//! its length (le32) and bytes. The node's executor writes it (`igneum/exec/src/day_stream.rs`), the miner fetches +//! it, the CLI's `--state` reads it, and a pack carries the leaves it yields as `leaves.bin`. + +use crate::blake2b::{blake2b_256, blake2b_512}; +use crate::seed::fnv1a64_words; + +pub const LEAF_TAG: &[u8] = b"igneum-sd1/"; +pub const SAMPLE_TAG: &[u8] = b"igneum-sd1-sample/"; +pub const STREAM_MAGIC: &[u8; 6] = b"IGSD1\0"; + +/// The canonical state stream at one chain block: what the executor serialises and what the leaves derive from. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct StateStream { + pub number: u64, + pub block: [u8; 32], + pub root: [u8; 32], + pub records: Vec>, +} + +impl StateStream { + pub fn encode(&self) -> Vec { + let mut b = Vec::with_capacity(6 + 8 + 32 + 32 + 4 + self.records.iter().map(|r| 4 + r.len()).sum::()); + b.extend_from_slice(STREAM_MAGIC); + b.extend_from_slice(&self.number.to_le_bytes()); + b.extend_from_slice(&self.block); + b.extend_from_slice(&self.root); + b.extend_from_slice(&(self.records.len() as u32).to_le_bytes()); + for r in &self.records { + b.extend_from_slice(&(r.len() as u32).to_le_bytes()); + b.extend_from_slice(r); + } + b + } + + pub fn decode(bytes: &[u8]) -> Result { + if bytes.len() < 6 + 8 + 32 + 32 + 4 || &bytes[..6] != STREAM_MAGIC { + return Err("not a state stream file (magic IGSD1)".into()); + } + let mut at = 6; + let number = u64::from_le_bytes(bytes[at..at + 8].try_into().unwrap()); + at += 8; + let block: [u8; 32] = bytes[at..at + 32].try_into().unwrap(); + at += 32; + let root: [u8; 32] = bytes[at..at + 32].try_into().unwrap(); + at += 32; + let n = u32::from_le_bytes(bytes[at..at + 4].try_into().unwrap()) as usize; + at += 4; + let mut records = Vec::with_capacity(n.min(1 << 20)); + for i in 0..n { + if at + 4 > bytes.len() { + return Err(format!("state stream truncated at record {i} of {n}")); + } + let len = u32::from_le_bytes(bytes[at..at + 4].try_into().unwrap()) as usize; + at += 4; + if at + len > bytes.len() { + return Err(format!("state stream truncated inside record {i} of {n}")); + } + records.push(bytes[at..at + len].to_vec()); + at += len; + } + if at != bytes.len() { + return Err(format!("state stream has {} trailing bytes", bytes.len() - at)); + } + Ok(StateStream { number, block, root, records }) + } + + pub fn read_file(path: &std::path::Path) -> Result { + let bytes = std::fs::read(path).map_err(|e| format!("read {}: {e}", path.display()))?; + Self::decode(&bytes) + } +} + +/// `D[i]`: the 64-byte digest of record `i` under `root`, as 16 little-endian words. +pub fn leaf_digest(root: &[u8; 32], i: u32, record: &[u8]) -> [u32; 16] { + let d = blake2b_512(&[LEAF_TAG, root, &i.to_le_bytes(), record]); + let mut w = [0u32; 16]; + for (k, x) in w.iter_mut().enumerate() { + *x = u32::from_le_bytes(d[k * 4..k * 4 + 4].try_into().unwrap()); + } + w +} + +/// The sample order key of a record under `root`. +pub fn sample_key(root: &[u8; 32], record: &[u8]) -> [u8; 32] { + blake2b_256(&[SAMPLE_TAG, root, record]) +} + +/// The leaves of one window (or day) of class v5: `n` digests of 64 bytes, `leaf(t) = D[t mod n]`. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct StateLeaves { + pub root: [u8; 32], + pub block: [u8; 32], + pub number: u64, + /// Records in the stream before any sample. + pub records_total: u64, + /// Whether the stream had more records than the dataset has items (the sample rule applied). + pub sampled: bool, + leaves: Vec<[u32; 16]>, +} + +impl StateLeaves { + /// The items a dataset of `2^log2_words` words has: `2^(log2_words - 4)`. + pub fn items_of(log2_words: u32) -> u64 { + 1u64 << log2_words.saturating_sub(4) + } + + /// The leaves of `records` (canonical order) under `root` for a dataset of `2^log2_words` words. An empty stream + /// yields one leaf, the digest of the empty record, so `n` is never 0. + pub fn build(root: [u8; 32], block: [u8; 32], number: u64, records: &[Vec], log2_words: u32) -> StateLeaves { + let items = Self::items_of(log2_words); + let records_total = records.len() as u64; + let empty: Vec> = vec![Vec::new()]; + let records = if records.is_empty() { &empty[..] } else { records }; + let sampled = records.len() as u64 > items; + let chosen: Vec<&Vec> = if sampled { + let mut keyed: Vec<([u8; 32], &Vec)> = records.iter().map(|r| (sample_key(&root, r), r)).collect(); + keyed.sort_unstable_by(|a, b| a.0.cmp(&b.0).then_with(|| a.1.cmp(b.1))); + keyed.into_iter().take(items as usize).map(|(_, r)| r).collect() + } else { + records.iter().collect() + }; + let leaves = chosen.iter().enumerate().map(|(i, r)| leaf_digest(&root, i as u32, r)).collect(); + StateLeaves { root, block, number, records_total, sampled, leaves } + } + + pub fn from_stream(s: &StateStream, log2_words: u32) -> StateLeaves { + Self::build(s.root, s.block, s.number, &s.records, log2_words) + } + + /// Leaves from the raw words of a `leaves.bin` (16 words per leaf), for a worker or a test that holds no stream. + pub fn from_words(root: [u8; 32], block: [u8; 32], number: u64, words: &[u32]) -> StateLeaves { + assert!(!words.is_empty() && words.len() % 16 == 0, "leaves are 16 words each"); + let leaves = words.chunks_exact(16).map(|c| c.try_into().unwrap()).collect::>(); + StateLeaves { root, block, number, records_total: leaves.len() as u64, sampled: false, leaves } + } + + #[inline(always)] + pub fn n(&self) -> u32 { + self.leaves.len() as u32 + } + + /// `leaf(t) = D[t mod n]`. + #[inline(always)] + pub fn leaf(&self, t: u32) -> &[u32; 16] { + &self.leaves[(t % self.n()) as usize] + } + + pub fn leaves(&self) -> &[[u32; 16]] { + &self.leaves + } + + /// The flat words of `leaves.bin`. + pub fn words(&self) -> Vec { + self.leaves.iter().flat_map(|l| l.iter().copied()).collect() + } + + /// The bytes of `leaves.bin` (little-endian words). + pub fn bytes(&self) -> Vec { + self.words().iter().flat_map(|w| w.to_le_bytes()).collect() + } + + /// FNV-1a 64 over the leaves as little-endian bytes (the pack's `IGNEUM_STATE_LEAVES_FNV64`). + pub fn fnv1a64(&self) -> u64 { + fnv1a64_words(&self.words()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::generator::{generate_class, V5_CLASS}; + use crate::memhard::{derive_item_leaves, Cache, MixParams, Shape}; + use crate::seed::day_key; + use crate::verify::{hash_warp, DatasetMode, DatasetSource}; + use std::sync::Arc; + + fn records(n: usize, salt: u8) -> Vec> { + (0..n).map(|i| vec![salt, i as u8, (i >> 8) as u8, 7]).collect() + } + + fn leaves(root: u8, n: usize, log2_words: u32) -> Arc { + Arc::new(StateLeaves::build([root; 32], [0x22; 32], 5, &records(n, root), log2_words)) + } + + /// The known-failed case, first: a hasher without the state (no leaves, the leaves of another root, the leaves + /// of a stream one record short, the previous window's leaves) is wrong on every item and every lane. + #[test] + fn a_stateless_hasher_is_wrong_on_every_item() { + let key = day_key("2026-10-03"); + let shape = Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: true }; + let cache = Arc::new(Cache::fill_log2(key, 16)); + let mp = MixParams::with_shape(key, shape); + let good = leaves(0x11, 93, 20); + let other_root = leaves(0x12, 93, 20); + let one_short = Arc::new(StateLeaves::build([0x13; 32], [0x22; 32], 5, &records(92, 0x11), 20)); // a record short means another root + let previous_window = leaves(0x10, 93, 20); + for (name, bad) in [("another root", other_root.clone()), ("one record short", one_short.clone()), ("the previous window", previous_window.clone())] { + let equal = (0..64u32).filter(|&t| derive_item_leaves(t * 7919, &mp, &cache, Some(&good)) == derive_item_leaves(t * 7919, &mp, &cache, Some(&bad))).count(); + assert_eq!(equal, 0, "{name}: {equal} of 64 items equal"); + } + let stateless = Shape { state: false, ..shape }; + let mp_stateless = MixParams::with_shape(key, stateless); + let equal = (0..64u32).filter(|&t| derive_item_leaves(t * 7919, &mp, &cache, Some(&good)) == derive_item_leaves(t * 7919, &mp_stateless, &cache, None)).count(); + assert_eq!(equal, 0, "no leaves at all: {equal} of 64 items equal"); + // the warp: a class v5 program over a small dataset, the same program and cache, other leaves + let program = generate_class("igneum-genesis", V5_CLASS); + let ds = DatasetSource::new_shape("2026-10-03", DatasetMode::MemoryHard, 20, shape).with_leaves(good.clone()); + let ds_other = DatasetSource::new_shape("2026-10-03", DatasetMode::MemoryHard, 20, shape).with_leaves(previous_window.clone()); + let a = hash_warp(&program, 0, &ds); + let b = hash_warp(&program, 0, &ds_other); + assert_eq!(a.iter().zip(b.iter()).filter(|(x, y)| x == y).count(), 0, "0 of 32 lanes agree"); + assert_eq!(hash_warp(&program, 0, &ds), a, "the same leaves hash the same"); + } + + /// Every item takes a leaf: `leaf(t) = D[t mod n]`, so items `t` and `t + n` share a leaf and still differ. + #[test] + fn every_item_is_keyed_and_the_leaf_wraps() { + let l = leaves(0x11, 93, 28); + assert_eq!(l.n(), 93); + assert!(!l.sampled); + assert_eq!(l.records_total, 93); + for t in [0u32, 1, 92, 93, 94, 1_000_000, u32::MAX] { + assert_eq!(l.leaf(t), l.leaf(t % 93)); + assert_eq!(*l.leaf(t), leaf_digest(&[0x11; 32], t % 93, &records(93, 0x11)[(t % 93) as usize])); + } + let key = day_key("2026-10-03"); + let shape = Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: true }; + let cache = Cache::fill_log2(key, 16); + let mp = MixParams::with_shape(key, shape); + assert_ne!(derive_item_leaves(5, &mp, &cache, Some(&l)), derive_item_leaves(5 + 93, &mp, &cache, Some(&l))); + // an empty stream yields one leaf (the digest of the empty record), never a division by zero + let empty = StateLeaves::build([0x11; 32], [0; 32], 0, &[], 28); + assert_eq!(empty.n(), 1); + assert_eq!(empty.records_total, 0); + assert_eq!(*empty.leaf(12_345), leaf_digest(&[0x11; 32], 0, &[])); + } + + /// Above the dataset size the records are sampled in the keyed order: a different root picks a different set, + /// and the set cannot be the first `items` records of the stream. + #[test] + fn the_sample_above_the_dataset_size_is_keyed_by_the_root() { + let recs = records(40, 0x33); + let a = StateLeaves::build([0x11; 32], [0; 32], 0, &recs, 8); + let b = StateLeaves::build([0x12; 32], [0; 32], 0, &recs, 8); + assert_eq!(StateLeaves::items_of(8), 16); + assert_eq!((a.n(), a.sampled, a.records_total), (16, true, 40)); + assert_ne!(a.leaves(), b.leaves(), "another root, another sample"); + // the positional first 16 are not the sample (with overwhelming probability for 40 choose 16) + let positional = StateLeaves::build([0x11; 32], [0; 32], 0, &recs[..16], 8); + assert_ne!(a.leaves(), positional.leaves()); + // the same inputs sample the same + assert_eq!(StateLeaves::build([0x11; 32], [0; 32], 0, &recs, 8), a); + // at the dataset size exactly, no sample + let c = StateLeaves::build([0x11; 32], [0; 32], 0, &recs[..16], 8); + assert!(!c.sampled && c.n() == 16); + } + + #[test] + fn stream_file_round_trip_and_refusals() { + let s = StateStream { number: 159_357, block: [0xaf; 32], root: [0x1c; 32], records: records(93, 1) }; + let bytes = s.encode(); + assert_eq!(&bytes[..6], STREAM_MAGIC); + assert_eq!(StateStream::decode(&bytes).unwrap(), s); + assert!(StateStream::decode(&bytes[..bytes.len() - 1]).is_err(), "truncated"); + let mut trailing = bytes.clone(); + trailing.push(0); + assert!(StateStream::decode(&trailing).is_err(), "trailing bytes"); + assert!(StateStream::decode(b"IGSD0\0").is_err(), "wrong magic"); + let l = StateLeaves::from_stream(&s, 28); + let back = StateLeaves::from_words(s.root, s.block, s.number, &l.words()); + assert_eq!(back.leaves(), l.leaves()); + assert_eq!(l.bytes().len(), 93 * 64); + assert_eq!(l.fnv1a64(), back.fnv1a64()); + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/src/verify.rs b/tools/attack/adv-accept-v5/igneum-pow/src/verify.rs new file mode 100644 index 000000000..9d1281aeb --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/src/verify.rs @@ -0,0 +1,935 @@ +//! The CPU reference interpreter for one 32-lane warp (`cpuWarpTraced` in the Swift) and the API the node +//! calls. Dataset words come from the memory-hard cache (default) or from the closed form (old packs). + +use crate::generator::{generate, generate_class, EraParams, Instr, LoadClass, Op, Program, ProgramClass, ITERATIONS, LANES}; +use crate::memhard::{hot_index, HotTable, Layout, MemhardCpu, Shape}; +use crate::seed::day_key; + +/// The load address of an era program (`docs/plans/era-layout.md` section 1.3): `y = rotl(x * M, R)`, then the +/// window of the load site, `k = min(win, D - 26)` (0 when `D <= 26`), `idx = ((y & (MASK >> k)) | ((off & +/// (2^k - 1)) << (D - k))) & MASK`. For every other class `idx = x & MASK`, the lottery hash's address. `mask` is +/// `2^D - 1`. The acceptance mirror calls this at the rule's constant `D = 28`. +#[inline(always)] +pub fn load_index(era: Option<&EraParams>, ins: &Instr, x: u32, mask: u32, log2: u32) -> u32 { + match era { + None => x & mask, + Some(e) => { + let (wm, off) = window(ins, mask, log2); + let y = x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot); + ((y & wm) | off) & mask + } + } +} + +/// The window of a load site at a dataset of `2^log2` words: `(window mask, offset)` such that +/// `idx = (y & window mask) | offset` lies in the site's aligned window of `2^(log2 - k)` words. +#[inline(always)] +pub fn window(ins: &Instr, mask: u32, log2: u32) -> (u32, u32) { + let k = (ins.win as u32).min(log2.saturating_sub(26)); + let wm = mask >> k; + let off = ((ins.off as u32) & ((1u32 << k) - 1)) << (log2 - k); + (wm, off) +} + +/// Read-width experiment (5 October 2026): a `load` of `W` words folds every word into `dst`: +/// `x = dst XOR w[0]; for j in 1..W: x = (rotl(x, FOLD_ROT) * FOLD_MUL) XOR w[j]; dst = x`. For `W = 1` this is the +/// lottery hash's `dst XOR dataset[...]`. The fold is state-dependent (the rotate-multiply sits between the words), +/// so no function of the line alone replaces it: two different lines give two different maps of `dst`, and a +/// dataset of folded lines cannot be stored in place of the dataset (see `docs/plans/read-width.md`). +pub const FOLD_ROT: u32 = 11; +pub const FOLD_MUL: u32 = 0x9E3779B1; + +/// The fold of `words` into `dst` (at least one word). +#[inline(always)] +pub fn fold_words(dst: u32, words: &[u32]) -> u32 { + let mut x = dst ^ words[0]; + for &w in &words[1..] { + x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w; + } + x +} + +/// Variant 5 (scratch): the fill value of word `j` (0..2) of slot `slot` of lane `lane` of the unit at base nonce +/// `base`, under program seed words `seed`. The scratch of a unit starts as these values; a slot written during +/// the unit's hash holds what was written. Mirrored as `scr_fill` in every emitted kernel. +#[inline(always)] +pub fn scratch_fill(seed: &[u32; 8], base: u32, lane: u32, slot: u32, j: u32) -> u32 { + splitmix32( + (base.wrapping_add(lane) ^ seed[j as usize]) + .wrapping_add(slot.wrapping_mul(0x9E3779B1)) + .wrapping_add((j + 1).wrapping_mul(0x85EBCA77)), + ) +} + +/// Variant 5: the 16-byte slot after a read-modify-write that read `w` and folded to `x`: `(x ^ w1, rotl(x, 7) ^ w2, +/// x + w0)` behind the slot's tag. +#[inline(always)] +pub fn scratch_rewrite(x: u32, w: &[u32; 3]) -> [u32; 3] { + [x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])] +} + +/// The CPU model of one unit's scratch (variant 5): per lane, the written slots and their words. Unwritten slots +/// read as [`scratch_fill`]. A unit touches at most `scratch ops x 32` slots; a GPU keeps the real scratch per +/// resident warp with a per-unit tag per slot. +pub struct ScratchModel { + slots: usize, + written: Vec, + data: Vec<[u32; 3]>, + pub reads: usize, + pub writes: usize, + /// Soundness tests (`tests/scratch.rs`, `docs/analysis/scratch-soundness.md`): when `Some`, every + /// read-modify-write is appended as it happened. `None` on every verification path. + pub trace: Option>, +} + +/// One scratch read-modify-write as the interpreter saw it (variant 5 soundness tests). +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ScratchEvent { + pub lane: u8, + pub slot: u32, + /// The slot had been written earlier in this unit (a re-hit): the words read were a rewrite, not the fill. + pub hit: bool, + pub read: [u32; 3], + /// The fold result, the new value of `dst`. + pub x: u32, + pub written: [u32; 3], +} + +impl ScratchModel { + pub fn new(slots_per_lane: usize) -> Self { + Self { + slots: slots_per_lane, + written: vec![false; LANES * slots_per_lane], + data: vec![[0; 3]; LANES * slots_per_lane], + reads: 0, + writes: 0, + trace: None, + } + } + /// Read slot `slot` of `lane`, then rewrite it from the fold result `x`. Returns the three words read. + #[inline] + pub fn rmw(&mut self, seed: &[u32; 8], base: u32, lane: usize, slot: u32, dst: u32) -> u32 { + let i = lane * self.slots + slot as usize; + let w = if self.written[i] { + self.data[i] + } else { + [ + scratch_fill(seed, base, lane as u32, slot, 0), + scratch_fill(seed, base, lane as u32, slot, 1), + scratch_fill(seed, base, lane as u32, slot, 2), + ] + }; + let x = fold_words(dst, &w); + let out = scratch_rewrite(x, &w); + if let Some(t) = self.trace.as_mut() { + t.push(ScratchEvent { lane: lane as u8, slot, hit: self.written[i], read: w, x, written: out }); + } + self.data[i] = out; + self.written[i] = true; + self.reads += 1; + self.writes += 1; + x + } +} + +/// Dataset element, closed form of (day words, index). The original prototype's six-operation element. +#[inline(always)] +pub fn dataset_elem(i: u32, d0: u32, d1: u32) -> u32 { + let mut x = i ^ d0; + x = x.wrapping_mul(0x9E3779B1); + x ^= x >> 15; + x = x.wrapping_add(d1); + x = x.wrapping_mul(0x85EBCA77); + x ^= x >> 13; + x = x.wrapping_mul(0xC2B2AE3D); + x ^= x >> 16; + x +} + +/// splitmix32, used for the register init. +#[inline(always)] +pub fn splitmix32(v: u32) -> u32 { + let mut x = v; + x ^= x >> 16; + x = x.wrapping_mul(0x7feb352d); + x ^= x >> 15; + x = x.wrapping_mul(0x846ca68b); + x ^= x >> 16; + x +} + +/// Which construction fills the dataset words. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum DatasetMode { + /// The six-operation closed form (packs igneum-genesis and igneum-hourly). Not memory-hard. + ClosedForm, + /// The 256 MiB cache and 8 dependent reads per item (pack igneum-genesis-mh, MEMHARD.md). The default. + MemoryHard, +} + +impl DatasetMode { + pub fn name(self) -> &'static str { + match self { + DatasetMode::ClosedForm => "closed-form", + DatasetMode::MemoryHard => "memory-hard", + } + } +} + +/// Where the interpreter reads dataset words from. +pub enum Dataset { + ClosedForm { d0: u32, d1: u32 }, + MemoryHard(MemhardCpu), +} + +/// A dataset of `2^log2` words plus the construction that fills it. +pub struct DatasetSource { + pub log2_words: u32, + pub mask: u32, + /// The day key `K`; `d0, d1 = K[0], K[1]`. + pub key: [u32; 8], + /// The bytes `K` was derived from (`"day/"` for a string day, `bind::day_bytes` on the chain), recorded + /// in packs so any implementation can rebuild the key. Empty when the key was given directly. + pub key_bytes: Vec, + pub dataset: Dataset, + /// The hot table of the epoch (hot-table experiment, `docs/plans/hot-table.md`): `Some` when the program's + /// class has one; filled by [`Epoch::new_class`] and [`Epoch::from_seed_bytes_class`] from the program's seed + /// bytes. A hot load reads `hot[hot_index(src, words)]`. + pub hot: Option, +} + +impl DatasetSource { + /// Build the source for a day. Memory-hard mode fills the 256 MiB cache on the calling thread. + pub fn new(day: &str, mode: DatasetMode, log2_words: u32) -> Self { + Self::new_shape(day, mode, log2_words, Shape::V2) + } + + /// [`DatasetSource::new`] with the construction's shape (mixer multiplier, cache size; Counter ASIC 2.0). + pub fn new_shape(day: &str, mode: DatasetMode, log2_words: u32, shape: Shape) -> Self { + let mut ds = Self::from_key_shape(day_key(day), mode, log2_words, shape); + ds.key_bytes = format!("day/{day}").into_bytes(); + ds + } + + pub fn from_key(key: [u32; 8], mode: DatasetMode, log2_words: u32) -> Self { + Self::from_key_shape(key, mode, log2_words, Shape::V2) + } + + /// [`DatasetSource::from_key`] with the construction's shape. Memory-hard mode fills a cache of + /// `2^shape.cache_log2_words` words on the calling thread. + pub fn from_key_shape(key: [u32; 8], mode: DatasetMode, log2_words: u32, shape: Shape) -> Self { + assert!((4..=32).contains(&log2_words), "dataset log2 must be in 4..=32"); + let mask = if log2_words == 32 { u32::MAX } else { (1u32 << log2_words) - 1 }; + let dataset = match mode { + DatasetMode::ClosedForm => Dataset::ClosedForm { d0: key[0], d1: key[1] }, + DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::with_shape(key, shape)), + }; + Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None } + } + + /// This source with the window's state leaves (class v5, `docs/design/class-v5-stored-state.md`): memory-hard mode + /// under a shape with `state` only. + pub fn with_leaves(mut self, leaves: std::sync::Arc) -> Self { + match &mut self.dataset { + Dataset::MemoryHard(m) => { + assert!(m.params.shape.state, "state leaves on a dataset whose shape has no state"); + m.leaves = Some(leaves); + } + Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"), + } + self + } + + /// A source of the same day with other leaves, the 256 MiB cache shared (the class v5 window refresh). + pub fn refreshed(&self, leaves: std::sync::Arc) -> Self { + let dataset = match &self.dataset { + Dataset::MemoryHard(m) => Dataset::MemoryHard(m.refreshed(leaves)), + Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"), + }; + Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } + } + + /// A copy of this source sharing its cache (and leaves), for a caller that needs an owned source from a shared one. + pub fn refreshed_or_clone(&self) -> Self { + let dataset = match &self.dataset { + Dataset::MemoryHard(m) => Dataset::MemoryHard(crate::memhard::MemhardCpu { params: m.params.clone(), cache: m.cache.clone(), leaves: m.leaves.clone() }), + Dataset::ClosedForm { d0, d1 } => Dataset::ClosedForm { d0: *d0, d1: *d1 }, + }; + Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None } + } + + /// The window's state leaves, when the source carries them. + pub fn leaves(&self) -> Option<&std::sync::Arc> { + self.memhard().and_then(|m| m.leaves.as_ref()) + } + + /// This source with the hot table of the epoch whose program seed bytes are `seed_bytes` (`mb` MiB). + pub fn with_hot(mut self, seed_bytes: &[u8], mb: u32) -> Self { + self.hot = Some(HotTable::for_seed_bytes(seed_bytes, mb)); + self + } + + /// The hot table of a program's class, filled from its seed bytes (none for a class without one). + pub fn attach_hot_for(&mut self, program: &Program) { + self.hot = program.class.hot.map(|h| HotTable::for_seed_bytes(&program.seed_bytes, h.mb as u32)); + } + + /// The shape of the memory-hard construction ([`Shape::V2`] for the closed form, which has none). + pub fn shape(&self) -> Shape { + self.memhard().map(|m| m.shape()).unwrap_or(Shape::V2) + } + + pub fn mode(&self) -> DatasetMode { + match self.dataset { + Dataset::ClosedForm { .. } => DatasetMode::ClosedForm, + Dataset::MemoryHard(_) => DatasetMode::MemoryHard, + } + } + + pub fn memhard(&self) -> Option<&MemhardCpu> { + match &self.dataset { + Dataset::MemoryHard(m) => Some(m), + _ => None, + } + } + + /// `dataset[w & mask]` under the linear layout (the lottery hash). + pub fn word(&self, w: u32) -> u32 { + self.word_at(Layout::LINEAR, w) + } + + /// `dataset[w & mask]` under a program's layout (era layout). The closed form has no items and ignores it. + pub fn word_at(&self, layout: Layout, w: u32) -> u32 { + let w = w & self.mask; + match &self.dataset { + Dataset::ClosedForm { d0, d1 } => dataset_elem(w, *d0, *d1), + Dataset::MemoryHard(m) => m.word_at(layout, w), + } + } + + /// `out[k] = dataset[idx[k]]`; indices are already masked. Returns items derived (0 for the closed form). + #[inline] + fn fetch(&self, idx: &[u32; LANES], out: &mut [u32; LANES], layout: Layout) -> usize { + match &self.dataset { + Dataset::ClosedForm { d0, d1 } => { + for k in 0..LANES { + out[k] = dataset_elem(idx[k], *d0, *d1); + } + 0 + } + Dataset::MemoryHard(m) => m.fetch(idx, out, layout), + } + } + + /// `out[k][j] = dataset[base[k] + j]` for `j < width`; bases are masked and aligned to `width` words + /// (`width` 4 or 16, so a lane's words lie in one item). Returns items derived (0 for the closed form). + #[inline] + fn fetch_wide(&self, base: &[u32; LANES], width: usize, out: &mut [[u32; 16]; LANES], layout: Layout) -> usize { + match &self.dataset { + Dataset::ClosedForm { d0, d1 } => { + for k in 0..LANES { + for j in 0..width { + out[k][j] = dataset_elem(base[k] + j as u32, *d0, *d1); + } + } + 0 + } + Dataset::MemoryHard(m) => m.fetch_wide(base, width, out, layout), + } + } +} + +/// The result of interpreting one warp. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct WarpResult { + pub hashes: [u64; LANES], + /// Distinct dataset items derived from the cache (0 in closed-form mode). + pub items_derived: usize, +} + +#[inline(always)] +fn mulhi32(a: u32, b: u32) -> u32 { + ((a as u64 * b as u64) >> 32) as u32 +} + +/// Interpret `program` for the 32 nonces `base_nonce .. base_nonce + 31` (wrapping). Registers are kept +/// register-major (`r[reg][lane]`) so the per-lane loops vectorise; the semantics are those of the Metal +/// and CUDA kernels instruction for instruction. +pub fn interpret_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) -> WarpResult { + interpret_warp_init(program, &program.seed, base_nonce, ds) +} + +/// [`interpret_warp`] with explicit init words `I` (section 1.6 of the spec). The packs use `I = program.seed`; +/// a block uses `I = bind::block_init_words(H, nonce)`. +pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, ds: &DatasetSource) -> WarpResult { + interpret_warp_scratch(program, seed, base_nonce, ds, false).0 +} + +/// [`interpret_warp_init`] that also returns every scratch read-modify-write of the unit in execution order +/// (lane-minor within an instruction, as the interpreter runs them) when `trace` is set; empty otherwise and for +/// a class without a scratch. For the soundness tests of variant 5 only. +pub fn interpret_warp_scratch( + program: &Program, + seed: &[u32; 8], + base_nonce: u32, + ds: &DatasetSource, + trace: bool, +) -> (WarpResult, Vec) { + let mask = ds.mask; + let log2 = ds.log2_words; + let era = program.class.era; + let layout = program.class.layout(); + let mut r = [[0u32; LANES]; 8]; + for lane in 0..LANES { + let nonce = base_nonce.wrapping_add(lane as u32); + for i in 0..8 { + let mut x = nonce ^ seed[i]; + x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1)); + x = splitmix32(x); + r[i][lane] = x ^ seed[(i + 1) & 7]; + } + } + let mut items_derived = 0usize; + let mut idx = [0u32; LANES]; + let mut val = [0u32; LANES]; + let mut scratch = if program.has_scratch() { Some(ScratchModel::new(program.class.scratch_slots_per_lane())) } else { None }; + if trace { + if let Some(m) = scratch.as_mut() { + m.trace = Some(Vec::new()); + } + } + let slot_mask = program.class.scratch_slot_mask(); + if program.has_hot() { + let h = ds.hot.as_ref().expect("a hot-table program needs the epoch's hot table on the dataset source"); + assert_eq!(h.n_words(), program.hot_words(), "the hot table's size is the class's"); + } + for _ in 0..ITERATIONS { + let sel = r[0]; + for ins in &program.instrs { + step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived); + if ins.op == Op::Scratch { + let m = scratch.as_mut().expect("a scratch op needs a scratch class"); + let (d, a) = (ins.dst as usize, ins.src as usize); + for lane in 0..LANES { + let slot = r[a][lane] & slot_mask; + r[d][lane] = m.rmw(&program.seed, base_nonce, lane, slot, r[d][lane]); + } + } + } + // Latency-shadow block (Counter ASIC 3.0 item 8): the block runs `reps` times after instruction 63 with the + // iteration's `sel`; it is empty on every class without a shadow, so version 2 and class v3 run nothing here. + for _ in 0..program.shadow_reps() { + for ins in &program.shadow { + step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived); + } + } + } + let mut hashes = [0u64; LANES]; + for lane in 0..LANES { + let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21); + let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27); + hashes[lane] = ((hi as u64) << 32) | lo as u64; + } + let events = scratch.and_then(|m| m.trace).unwrap_or_default(); + (WarpResult { hashes, items_derived }, events) +} + +#[inline(always)] +#[allow(clippy::too_many_arguments)] +fn step( + ins: &Instr, + r: &mut [[u32; LANES]; 8], + sel: &[u32; LANES], + mask: u32, + log2: u32, + era: Option<&EraParams>, + layout: Layout, + ds: &DatasetSource, + idx: &mut [u32; LANES], + val: &mut [u32; LANES], + items_derived: &mut usize, +) { + let d = ins.dst as usize; + let a = ins.src as usize; + match ins.op { + Op::Add => { + let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32); + let src = r[a]; + for lane in 0..LANES { + let s = (sel[lane] >> bit) & 1; + let c = if s != 0 { imm2 } else { imm }; + r[d][lane] = r[d][lane].wrapping_add(src[lane]).wrapping_add(c); + } + } + Op::Sub => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].wrapping_sub(src[lane]); + } + } + Op::Mul => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].wrapping_mul(src[lane]); + } + } + Op::MulHi => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = mulhi32(r[d][lane], src[lane]); + } + } + Op::Xor => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] ^= src[lane]; + } + } + Op::Or => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] |= src[lane]; + } + } + Op::Rotl => { + let n = ins.rot; + for lane in 0..LANES { + r[d][lane] = r[d][lane].rotate_left(n); + } + } + Op::Rotr => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].rotate_right(src[lane] & 31); + } + } + Op::Mad => { + let src = r[a]; + let src2 = r[ins.src2 as usize]; + for lane in 0..LANES { + r[d][lane] = src[lane].wrapping_mul(src2[lane]).wrapping_add(r[d][lane]); + } + } + Op::Shfl => { + let src = r[a]; + let m = ins.mask as usize; + for lane in 0..LANES { + r[d][lane] ^= src[lane ^ m]; + } + } + Op::Load if ins.width == 1 => { + for lane in 0..LANES { + idx[lane] = load_index(era, ins, r[a][lane], mask, log2); + } + *items_derived += ds.fetch(idx, val, layout); + for lane in 0..LANES { + r[d][lane] ^= val[lane]; + } + } + Op::Load => { + // Read-width experiment: `width` words from the aligned address, every word folded into dst. + let width = ins.width as usize; + let align = !(ins.width as u32 - 1); + for lane in 0..LANES { + idx[lane] = load_index(era, ins, r[a][lane], mask, log2) & align; + } + let mut vals = [[0u32; 16]; LANES]; + *items_derived += ds.fetch_wide(idx, width, &mut vals, layout); + for lane in 0..LANES { + r[d][lane] = fold_words(r[d][lane], &vals[lane][..width]); + } + } + Op::Scratch => { + // handled by the caller (interpret_warp_init), which owns the unit's scratch model + } + Op::Hot => { + // Hot-table experiment: one word of the epoch table at the multiply-shift index, plain xor fold. + let h = ds.hot.as_ref().expect("a hot load needs the hot table"); + let n = h.n_words(); + for lane in 0..LANES { + r[d][lane] ^= h.at(hot_index(r[a][lane], n)); + } + } + Op::WLoad => { + // Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l. + let base = (r[a][0] & mask) & !31; + for lane in 0..LANES { + idx[lane] = base + lane as u32; + } + *items_derived += ds.fetch(idx, val, Layout::LINEAR); + for lane in 0..LANES { + r[d][lane] ^= val[lane]; + } + } + } +} + +/// The 32 hashes of one warp (`cpuWarp` in the Swift). +pub fn hash_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) -> [u64; LANES] { + interpret_warp(program, base_nonce, ds).hashes +} + +/// Everything a node needs to verify blocks of one epoch on one day: the program for the epoch seed and the +/// dataset source for the day key. Building one in memory-hard mode fills the 256 MiB cache (about 0.2 s on +/// one core); keep it for the whole epoch and share it between threads (`&Epoch` is `Send + Sync`). +pub struct Epoch { + pub program: Program, + pub dataset: DatasetSource, +} + +/// Default dataset size: 2^28 words = 1 GiB. +pub const DEFAULT_DATASET_LOG2: u32 = 28; + +/// Days a day index lies after the network's genesis day (0 for the genesis day and any day before it). The node's +/// entry; the same function as `memhard::days_since_genesis`. +pub fn days_since_genesis(day_index: u64, genesis_day_index: u64) -> u64 { + crate::memhard::days_since_genesis(day_index, genesis_day_index) +} + +impl Epoch { + pub fn new(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32) -> Self { + Self { program: generate(seed), dataset: DatasetSource::new(day, mode, dataset_log2) } + } + + /// [`Epoch::new`] with a load class (read-width experiment; Counter ASIC 2.0: the class's mixer multiplier + /// shapes the dataset, the cache is the genesis size since a string day has no day index). + pub fn new_class(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32, class: LoadClass) -> Self { + Self::new_class_day(seed, day, mode, dataset_log2, class, 0) + } + + /// [`Epoch::new_class`] on day `days_since_genesis` of the growth schedule (the cache of + /// `memhard::cache_log2_words` for a class with the growth rule; the dataset size is the caller's). + pub fn new_class_day(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32, class: LoadClass, days_since_genesis: u64) -> Self { + let shape = Shape::for_class_day(&class, days_since_genesis); + let program = generate_class(seed, class); + let mut dataset = DatasetSource::new_shape(day, mode, dataset_log2, shape); + // hot-table experiment: a hot class fills its table from the seed bytes + dataset.attach_hot_for(&program); + Self { program, dataset } + } + + /// `dataset[w]` as this epoch's program reads it: under the program's layout (era layout; linear for v2). + pub fn dataset_word(&self, w: u32) -> u32 { + self.dataset.word_at(self.program.class.layout(), w) + } + + /// The production shape: memory-hard, 1 GiB dataset. + pub fn memory_hard(seed: &str, day: &str) -> Self { + Self::new(seed, day, DatasetMode::MemoryHard, DEFAULT_DATASET_LOG2) + } + + /// The chain's shape: program from the 32-byte epoch seed (devnet: the epoch block hash; later the VDF + /// output) through the version 2 generator and its acceptance rule, and the cache from + /// `seed_words_from_bytes(day_bytes)` (`bind::day_bytes`). Memory-hard, 1 GiB dataset. `label` is only + /// recorded in emitted packs. + pub fn from_seed_bytes(epoch_seed: &[u8], day_bytes: &[u8], label: &str) -> Self { + Self::from_seed_bytes_class(epoch_seed, day_bytes, label, LoadClass::V2) + } + + /// [`Epoch::from_seed_bytes`] with a load class (read-width experiment; Counter ASIC 2.0: the class's mixer + /// multiplier shapes the dataset). Day 0 of the growth schedule: the 2^26-word cache and the 2^28-word dataset, + /// which is every devnet pack and vector. A node past the first doubling calls [`Epoch::from_seed_bytes_day`]. + pub fn from_seed_bytes_class(epoch_seed: &[u8], day_bytes: &[u8], label: &str, class: LoadClass) -> Self { + Self::from_seed_bytes_day(epoch_seed, day_bytes, label, class, 0, DEFAULT_DATASET_LOG2) + } + + /// The chain's shape on day `days_since_genesis` (`memhard::days_since_genesis(day_index(header), day_index(genesis))`, + /// the node's two day indices): the program of the class, and under the class's growth rule the cache of + /// `memhard::cache_log2_words(d)` and the dataset of `memhard::dataset_log2_words(genesis_dataset_log2, d)` + /// (the genesis size is 28 for the 1 GiB devnet, 29 for the designed 2 GiB). Without the growth rule the cache + /// is 2^26 words and the dataset `2^genesis_dataset_log2` on every day. + pub fn from_seed_bytes_day(epoch_seed: &[u8], day_bytes: &[u8], label: &str, class: LoadClass, days_since_genesis: u64, genesis_dataset_log2: u32) -> Self { + let program = crate::generator::generate_from_seed_bytes_class(label, epoch_seed, class); + let key = crate::seed::seed_words_from_bytes(day_bytes); + let shape = Shape::for_class_day(&class, days_since_genesis); + let dataset_log2 = if class.growth { crate::memhard::dataset_log2_words(genesis_dataset_log2, days_since_genesis) } else { genesis_dataset_log2 }; + let mut dataset = DatasetSource::from_key_shape(key, DatasetMode::MemoryHard, dataset_log2, shape); + dataset.key_bytes = day_bytes.to_vec(); + dataset.attach_hot_for(&program); + Self { program, dataset } + } + + /// The chain's shape with the program class (Counter ASIC 2.0, 5 October 2026): what the node's engine and the + /// miner's pack export build from the seeds a block template carries. Class v2 is [`Epoch::from_seed_bytes`] + /// exactly (the era bytes are ignored and not recorded); class v3 draws from [`crate::generator::V3_CLASS`] + /// with generator version 3 and records the era seed bytes (`E_n`) in the program for the pack. + pub fn from_chain_seeds(epoch_seed: &[u8], day_bytes: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, label: &str) -> Self { + Self { program: Self::chain_program(epoch_seed, era_bytes, class, label), dataset: Self::chain_dataset(day_bytes, class) } + } + + /// The program alone of [`Epoch::from_chain_seeds`] (no cache fill): for an engine that shares the day's cache. + pub fn chain_program(epoch_seed: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, label: &str) -> Program { + crate::generator::generate_from_seed_bytes_program_class(label, epoch_seed, class, era_bytes) + } + + /// [`Epoch::chain_program`] at a rung of the latency ladder (`docs/design/latency-ladder.md`): `shadow_reps` is + /// the shadow pass count the chain's step gives the epoch (0 = the class's own, which is [`Epoch::chain_program`] + /// byte for byte). The node's engine and the miner's pack export call this with the step the template names. + pub fn chain_program_shadow(epoch_seed: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, shadow_reps: u16, label: &str) -> Program { + crate::generator::generate_from_seed_bytes_program_class_shadow(label, epoch_seed, class, era_bytes, shadow_reps) + } + + /// The day's cache and dataset of [`Epoch::from_chain_seeds`], the one entry the node's engine builds a day + /// cache through. The class is an argument because the Counter ASIC 2.0 integration gives class v3 its own item + /// construction (the mixer multiplier) and cache size schedule (ca2-mixer); today both classes build the day of + /// [`Epoch::from_seed_bytes`], and the engine keys its day caches on `(day, class)` so the two never share one. + pub fn chain_dataset(day_bytes: &[u8], class: ProgramClass) -> DatasetSource { + Self::chain_dataset_day(day_bytes, class, 0, DEFAULT_DATASET_LOG2) + } + + /// [`Epoch::chain_dataset`] with the day's position since genesis and the network's genesis dataset size: the + /// entry the node's engine and the miner's export build every day cache through, so the cache growth schedule + /// of spec 01 section 1.13.3 has one place to act (ca2-mixer, 5 October 2026, `docs/plans/mixer-x4.md`): the + /// class's load class gives the mixer multiplier and whether the growth rule applies (`Shape::for_class_day`); + /// under the rule the cache is `2^memhard::cache_log2_words(d)` words and the dataset + /// `2^memhard::dataset_log2_words(genesis_dataset_log2, d)`; without it (class v2) the cache is 2^26 words and + /// the dataset the genesis size on every day. `days_since_genesis` is [`days_since_genesis`] of the block's and + /// the genesis header's day indices. + pub fn chain_dataset_day(day_bytes: &[u8], class: ProgramClass, days_since_genesis: u64, genesis_dataset_log2: u32) -> DatasetSource { + let lc = class.load_class(); + let shape = Shape::for_class_day(&lc, days_since_genesis); + let dataset_log2 = if lc.growth { crate::memhard::dataset_log2_words(genesis_dataset_log2, days_since_genesis) } else { genesis_dataset_log2 }; + let key = crate::seed::seed_words_from_bytes(day_bytes); + let mut dataset = DatasetSource::from_key_shape(key, DatasetMode::MemoryHard, dataset_log2, shape); + dataset.key_bytes = day_bytes.to_vec(); + dataset + } + + /// The 32 hashes of the warp starting at `base_nonce`. + pub fn hash_warp(&self, base_nonce: u32) -> [u64; LANES] { + hash_warp(&self.program, base_nonce, &self.dataset) + } + + pub fn interpret_warp(&self, base_nonce: u32) -> WarpResult { + interpret_warp(&self.program, base_nonce, &self.dataset) + } + + /// The hash of one nonce. The verification unit is a warp, so the 31 sibling nonces of the aligned + /// 32-nonce group are computed too (the shuffles couple the lanes). + pub fn hash(&self, nonce: u32) -> u64 { + self.hash_warp(nonce & !31)[(nonce & 31) as usize] + } + + /// `hash(nonce) <= target`. The hash is 64 bits; the fork maps it into its 256-bit target space. + pub fn verify_block(&self, nonce: u32, target: u64) -> bool { + self.hash(nonce) <= target + } +} + +/// One-shot `verify_block(seed, nonce, target)`: builds the epoch (cache fill included) and checks. For a +/// node use [`Epoch`] and keep it; this exists for scripts and tests. +pub fn verify_block(seed: &str, day: &str, nonce: u32, target: u64) -> bool { + Epoch::memory_hard(seed, day).verify_block(nonce, target) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn closed_form_head_matches_pack() { + // vectors.json dataset_head for igneum-genesis (closed form), day 2026-10-03. + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28); + assert_eq!(ds.word(0), 0x82174c0f); + assert_eq!(ds.word(1), 0x577bdb9c); + assert_eq!(ds.word(0x0fffffff), 0xf78c84a4); + } + + /// Read-width experiment: the fold for one word is a plain xor; a wide fetch hands each lane the words the + /// scalar path would; two distinct lines give two distinct maps of dst (one point suffices as a smoke check). + #[test] + fn fold_and_wide_fetch() { + assert_eq!(fold_words(0x1234_5678, &[0xdead_beef]), 0x1234_5678 ^ 0xdead_beef); + let w = [1u32, 2, 3, 4]; + let x = fold_words(7, &w); + let mut y: u32 = 7 ^ 1; + for &v in &w[1..] { + y = y.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ v; + } + assert_eq!(x, y); + assert_ne!(fold_words(7, &[1, 2, 3, 4]), fold_words(7, &[1, 2, 3, 5])); + let ds = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 20); + let mut base = [0u32; LANES]; + for (k, b) in base.iter_mut().enumerate() { + *b = ((k as u32).wrapping_mul(0x9E37_79B1) & ds.mask) & !15; + } + let mut out = [[0u32; 16]; LANES]; + let items = ds.fetch_wide(&base, 16, &mut out, Layout::LINEAR); + assert!(items >= 1 && items <= LANES); + for k in 0..LANES { + for j in 0..16 { + assert_eq!(out[k][j], ds.word(base[k] + j as u32), "lane {k} word {j}"); + } + } + let mut base4 = base; + for b in base4.iter_mut() { + *b += 8; + } + let items4 = ds.fetch_wide(&base4, 4, &mut out, Layout::LINEAR); + assert_eq!(items4, items); + for k in 0..LANES { + for j in 0..4 { + assert_eq!(out[k][j], ds.word(base4[k] + j as u32)); + } + } + } + + /// A wide-load program interprets identically on the closed form and through the memory-hard path's fold + /// (the same fold code), and a mixed-class epoch builds and hashes. + /// Variant 5: a fill word is deterministic, a rewrite changes the slot, and a second read of a written slot + /// returns the rewrite, not the fill. + #[test] + fn scratch_model() { + let seed = [1u32, 2, 3, 4, 5, 6, 7, 8]; + assert_eq!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 1)); + assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2)); + assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 64, 3, 100, 1)); + let mut m = ScratchModel::new(256); + let w = [scratch_fill(&seed, 32, 3, 100, 0), scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2)]; + let x = m.rmw(&seed, 32, 3, 100, 0xabcd); + assert_eq!(x, fold_words(0xabcd, &w)); + let x2 = m.rmw(&seed, 32, 3, 100, 0xabcd); + assert_eq!(x2, fold_words(0xabcd, &scratch_rewrite(x, &w))); + assert_eq!(m.reads, 2); + let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::scratch(4, 128)); + assert_eq!(e.program.scratch_ops_per_hash(), 32); + assert_eq!(e.hash_warp(0), e.hash_warp(0)); + } + + /// Era layout: the load address stays inside the site's window and below the mask at every dataset size (the + /// window floor of 2^26 words clamps the shrink), the interleaved memory-hard dataset reads item(t(w))[j(w)] and + /// is the same prefix at 2^20 and 2^22 words, the wide fetch agrees word for word, and an era epoch hashes + /// deterministically through the interpreter and the single-nonce API. + #[test] + fn era_windows_layout_and_epochs() { + let eb = EraParams::test_era_bytes("igneum-era-test/1"); + let c = LoadClass::era(LoadClass::V2, &eb, &[1]); + let e = c.era.unwrap(); + let mut ins = Instr { op: Op::Load, dst: 0, src: 1, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 2, off: 3 }; + let mut s = crate::seed::SplitMix64::new(7); + for log2 in [20u32, 26, 27, 28, 29] { + let mask = (1u64 << log2) as u32 - 1; + let k = ins.win.min(log2.saturating_sub(26) as u8) as u32; + for _ in 0..1000 { + let x = s.next() as u32; + let idx = load_index(Some(&e), &ins, x, mask, log2); + assert!(idx <= mask); + let (wm, off) = window(&ins, mask, log2); + assert_eq!(idx & !wm, off, "log2 {log2}"); + assert_eq!(wm, mask >> k); + assert_eq!(idx, ((x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot) & wm) | off) & mask); + } + } + ins.win = 0; + assert_eq!(load_index(None, &ins, 0xdead_beef, 0x0fff_ffff, 28), 0xdead_beef & 0x0fff_ffff); + // the interleaved dataset: one day cache, the layout per program + let l = e.layout(); + assert_eq!(l.pos, [1, 3, 8, 13]); + let small = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 20); + let big = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 22); + let m = small.memhard().unwrap(); + for w in [0u32, 1, 4, 5, 255, 256, 4095, 8192, 0x0f_ffff] { + let (t, j) = l.split(w); + assert_eq!(small.word_at(l, w), crate::memhard::derive_item(t, &m.params, &m.cache)[j as usize], "w {w}"); + assert_eq!(small.word_at(l, w), big.word_at(l, w), "prefix at w {w}"); + assert_eq!(small.word(w), small.word_at(Layout::LINEAR, w)); + } + let mut idx = [0u32; LANES]; + for (k, i) in idx.iter_mut().enumerate() { + *i = (k as u32).wrapping_mul(0x9E37_79B1) & small.mask; + } + let mut out = [0u32; LANES]; + small.fetch(&idx, &mut out, l); + for k in 0..LANES { + assert_eq!(out[k], small.word_at(l, idx[k])); + } + // a 16-byte era: the wide fetch keeps a lane's four words in one item + let eb3 = EraParams::test_era_bytes("igneum-era-test/3"); + let c3 = LoadClass::era(LoadClass::fixed(4, 16), &eb3, &[4]); + let l3 = c3.layout(); + assert_eq!(l3.pos[..2], [0, 1]); + let mut base = [0u32; LANES]; + for (k, b) in base.iter_mut().enumerate() { + *b = ((k as u32).wrapping_mul(0x9E37_79B1) & small.mask) & !3; + } + let mut wide = [[0u32; 16]; LANES]; + small.fetch_wide(&base, 4, &mut wide, l3); + for k in 0..LANES { + for j in 0..4 { + assert_eq!(wide[k][j], small.word_at(l3, base[k] + j as u32), "lane {k} word {j}"); + } + } + // era epochs hash deterministically, differ per era, and the single-nonce API agrees with the warp + let mut seen = std::collections::HashSet::new(); + for (n, c) in [(1u64, c), (3, c3)] { + let ep = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::MemoryHard, 20, c); + assert_eq!(ep.dataset_word(5), ep.dataset.word_at(c.layout(), 5)); + let a = ep.hash_warp(64); + assert_eq!(a, ep.hash_warp(64)); + assert_eq!(ep.hash(64 + 5), a[5]); + assert!(seen.insert(a[0]), "era {n}"); + let closed = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, c); + assert_ne!(closed.hash_warp(64), a); + } + } + + /// Hot-table experiment: an epoch of a hot class carries the table, hashes deterministically and differs from + /// version 2; the reference interpreter agrees with a hand-stepped hot load; a hot program without its table is + /// refused. + #[test] + fn hot_epochs_hash() { + let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(32, 4)); + let h = e.dataset.hot.as_ref().expect("the epoch fills the hot table"); + assert_eq!(h.n_words(), 1 << 23); + assert_eq!(h.key, crate::memhard::hot_key(b"igneum-genesis")); + let v2 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::V2); + let a = e.hash_warp(0); + assert_eq!(a, e.hash_warp(0)); + assert_ne!(a, v2.hash_warp(0)); + assert_ne!(a[0], a[1]); + // the same program under a 64 MiB table reads other words + let e64 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(64, 4)); + assert_eq!(e64.program.instrs, e.program.instrs); + assert_ne!(e64.hash_warp(0), a); + // from seed bytes, the chain's shape, with a hot class + let genesis = crate::bind::unhex("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07").unwrap(); + let ec = Epoch::from_seed_bytes_class(&genesis, &crate::bind::day_bytes(20_730), "devnet", LoadClass::hot(32, 2)); + assert_eq!(ec.dataset.hot.as_ref().unwrap().key, crate::memhard::hot_key(&genesis)); + assert_eq!(ec.hash_warp(0), ec.hash_warp(0)); + } + + #[test] + #[should_panic(expected = "needs the epoch's hot table")] + fn hot_program_without_a_table_is_refused() { + let p = generate_class("igneum-genesis", LoadClass::hot(32, 4)); + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 20); + let _ = hash_warp(&p, 0, &ds); + } + + #[test] + fn wide_class_epochs_hash() { + for name in ["w16", "w64x4", "50,35,15"] { + let c = LoadClass::parse(name).unwrap(); + let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, c); + assert_eq!(e.program.class, c); + let a = e.hash_warp(0); + let b = e.hash_warp(0); + assert_eq!(a, b); + assert_ne!(a[0], a[1]); + } + } + + #[test] + fn closed_form_genesis_vector_lane0() { + // Generator v2 vectors (4 October 2026), proto-cuda/packs/igneum-genesis/vectors.json. + let e = Epoch::new("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 28); + let w = e.hash_warp(0); + assert_eq!(w[0], 0x31e7555c3dfd007f); + assert_eq!(w[31], 0xaab617183923ab2a); + assert_eq!(e.hash(0), w[0]); + assert_eq!(e.hash(31), w[31]); + assert!(e.verify_block(0, u64::MAX)); + assert!(e.verify_block(0, w[0])); + assert!(!e.verify_block(0, w[0] - 1)); + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/tests/derive.rs b/tools/attack/adv-accept-v5/igneum-pow/tests/derive.rs new file mode 100644 index 000000000..a64af0292 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/tests/derive.rs @@ -0,0 +1,275 @@ +//! The per-day item-derivation program (Counter ASIC 3.0 item 2, `docs/plans/counter-asic-3-derivation.md`): the +//! soundness runs on the CPU. +//! +//! 1. By hand: the derived item of class dr736 restated with the scalar reference on a small cache equals the +//! verifier's batched (SoA) derivation, at the item boundary and the index wrap. +//! 2. The v2 and v3 paths are untouched: the derivation field is 0 on both, their items are the fixed-mixer items. +//! 3. Determinism and the stream: two epochs agree on every vector and file; the program is the continuation of the +//! mixer's stream (the first draw of the program follows the 40 mixer draws); a different day draws a different +//! program; the class is in the program id and the name. +//! 4. Stats beside x8: bit balance and single-bit avalanche of the derived items and of the hash, on the same seeds. +//! 5. A program whose text is the kernels': every instruction's C text evaluated by hand on one state matches the +//! scalar reference (the text forms are what Metal, CUDA and OpenCL compile). + +use igneum_pow::derive::{instr_text, run_round_scalar, DOp, DeriveProgram, DERIVE_LEN_X8, DERIVE_PROGRAMS}; +use igneum_pow::emit::export_pack; +use igneum_pow::generator::{generate_from_seed_bytes_class, LoadClass, V3_CLASS}; +use igneum_pow::memhard::{derive_item, derive_items, mixer, round_key, Cache, MixParams, Shape, ITEM_ROUNDS}; +use igneum_pow::seed::{day_key, SplitMix64}; +use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch}; + +const DAY: &str = "2026-10-03"; + +/// The item of a derivation class restated by hand with the scalar reference. +fn item_by_hand(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] { + let prog = mp.derive.as_ref().expect("a derivation class"); + let mut s = [0u32; 16]; + s[..8].copy_from_slice(&mp.key); + for i in 0..8 { + s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); + } + for r in 0..ITEM_ROUNDS { + run_round_scalar(&prog.rounds[r], &mut s); + let line = cache.line(s[0]); + for i in 0..16 { + s[i] ^= line[i]; + } + } + run_round_scalar(&prog.rounds[ITEM_ROUNDS], &mut s); + s +} + +#[test] +fn derived_item_by_hand_and_in_batches() { + let key = day_key(DAY); + let cache = Cache::fill_log2(key, 16); + let shape = Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false }; + let mp = MixParams::with_shape(key, shape); + let prog = mp.derive.as_ref().unwrap(); + assert_eq!(prog.rounds.len(), DERIVE_PROGRAMS); + assert!(prog.check().is_ok()); + assert_eq!(shape.derive_instrs_per_item(), 9 * DERIVE_LEN_X8); + assert_eq!(shape.mixers_per_item(), 0); + for t in [0u32, 1, 2, 15, 16, 17, 12_345, (1 << 28) - 1, u32::MAX - 1, u32::MAX] { + assert_eq!(derive_item(t, &mp, &cache), item_by_hand(t, &mp, &cache), "t {t}"); + } + // a batch of 32 distinct items against one at a time, and a short batch + let ts: Vec = (0..32).map(|k| k * 7_919 + 3).collect(); + let mut out = [[0u32; 16]; 32]; + derive_items(&ts, &mp, &cache, &mut out); + for (k, &t) in ts.iter().enumerate() { + assert_eq!(out[k], item_by_hand(t, &mp, &cache), "batch slot {k}"); + } + let mut out5 = [[0u32; 16]; 5]; + derive_items(&ts[..5], &mp, &cache, &mut out5); + assert_eq!(&out5[..], &out[..5]); + // the fixed mixer of the same key gives other items + let v3 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: false }); + assert!(v3.derive.is_none()); + assert_ne!(derive_item(0, &v3, &cache), derive_item(0, &mp, &cache)); +} + +#[test] +fn v2_and_v3_are_untouched() { + assert_eq!(LoadClass::V2.derive_len, 0); + assert_eq!(V3_CLASS.derive_len, 0); + assert_eq!(LoadClass::MX8.derive_len, 0); + assert_eq!(Shape::V2.derive_len, 0); + assert!(!Shape::for_class(&V3_CLASS).is_derived()); + let key = day_key(DAY); + let cache = Cache::fill_log2(key, 16); + // the version 2 item restated by hand (the mixer_mult_by_hand test of memhard.rs, m = 1) + let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0, state: false }); + let t = 12_345u32; + let mut s = [0u32; 16]; + s[..8].copy_from_slice(&key); + for i in 0..8 { + s[8 + i] = t.wrapping_mul(v2.mul[i]).wrapping_add(v2.rc[i]); + } + for r in 0..8usize { + mixer(&mut s, round_key(r), &v2); + let line = cache.line(s[0]); + for i in 0..16 { + s[i] ^= line[i]; + } + } + mixer(&mut s, round_key(8), &v2); + assert_eq!(derive_item(t, &v2, &cache), s); + // the mixer constants of the derivation class are the v2 draws (the stream continues after them) + let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false }); + assert_eq!((dr.rot, dr.mul, dr.rc), (v2.rot, v2.mul, v2.rc)); +} + +#[test] +fn stream_class_name_and_id() { + let key = day_key(DAY); + // the program is the continuation of the mixer stream: 40 draws, then the program + let mut rng = SplitMix64::new(key[0] as u64 | ((key[1] as u64) << 32)); + for _ in 0..40 { + rng.next(); + } + let expect = DeriveProgram::draw(&mut rng, DERIVE_LEN_X8); + let mp = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8, state: false }); + assert_eq!(mp.derive.as_ref().unwrap(), &expect); + // another day, another program; another length, another program + let other = MixParams::with_shape(day_key("2026-10-04"), Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8, state: false }); + assert_ne!(other.derive.as_ref().unwrap().fingerprint(), expect.fingerprint()); + let short = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: 368, state: false }); + assert_eq!(short.derive.as_ref().unwrap().instr_count(), 9 * 368); + // the class: name, parse, id, and the v2 program stream (v2 loads, no width roll) + let c = LoadClass::DR736; + assert_eq!(c.name(), "dr736"); + assert_eq!(LoadClass::parse("dr736"), Some(c)); + assert_eq!(LoadClass::parse("dr368"), Some(LoadClass::MX4.with_derive(368))); + assert_eq!(LoadClass::parse("dr0"), None); + assert_eq!(LoadClass::parse("dr5000"), None); + assert!(c.v2_loads() && !c.takes_width_roll() && c.growth && c.mixer_mult == 1 && c.is_derived()); + let p = generate_from_seed_bytes_class("igneum-genesis", b"igneum-genesis", c); + let v2 = generate_from_seed_bytes_class("igneum-genesis", b"igneum-genesis", LoadClass::V2); + let mx8 = generate_from_seed_bytes_class("igneum-genesis", b"igneum-genesis", LoadClass::MX8); + assert_eq!(p.instrs, v2.instrs, "the v2 program of the seed"); + assert_ne!(p.program_id(), v2.program_id()); + assert_ne!(p.program_id(), mx8.program_id()); + assert_ne!( + generate_from_seed_bytes_class("igneum-genesis", b"igneum-genesis", LoadClass::MX4.with_derive(368)).program_id(), + p.program_id() + ); +} + +#[test] +fn determinism_and_pack_text() { + let a = Epoch::new_class("igneum-genesis", DAY, DatasetMode::MemoryHard, 20, LoadClass::DR736); + let b = Epoch::new_class("igneum-genesis", DAY, DatasetMode::MemoryHard, 20, LoadClass::DR736); + let pa = export_pack(&a, DAY, "test"); + let pb = export_pack(&b, DAY, "test"); + assert_eq!(pa.outs, pb.outs); + assert_eq!(pa.files, pb.files); + let names: Vec<&str> = pa.files.iter().map(|(n, _)| n.as_str()).collect(); + assert!(names.contains(&"memhard.h") && names.contains(&"memhard.metal")); + for (name, text) in &pa.files { + if name == "memhard.h" || name == "memhard.metal" || name == "kernel.cl" { + for r in 0..DERIVE_PROGRAMS { + assert!(text.contains(&format!("mh_round_{r}(")), "{name} carries round program {r}"); + } + assert!(!text.contains("mh_mixer(s,"), "{name}: no mixer application under a derivation program"); + let dp = a.dataset.memhard().unwrap().params.derive.as_ref().unwrap(); + // every instruction's text appears, in order, inside the round functions + let first = instr_text(&dp.rounds[0][0]); + assert!(text.contains(&first), "{name} carries the first instruction {first}"); + } + if name == "program.h" { + assert!(text.contains("#define IGNEUM_LOAD_CLASS \"dr736\"")); + assert!(text.contains("#define IGNEUM_DERIVE_LEN 736")); + assert!(text.contains("#define IGNEUM_CLASS_DERIVE_LEN 736")); + assert!(!text.contains("#define IGNEUM_MIXER_MULT")); + assert!(text.contains("#define IGNEUM_GENERATOR 2")); + } + if name == "program.json" { + assert!(text.contains("\"derive_len\": 736")); + assert!(text.contains("\"programs\": [")); + serde_json::from_str::(text).expect("valid JSON"); + } + } + // the vectors are the interpreter's + assert_eq!(pa.outs[0], a.hash_warp(0)); +} + +/// Bit balance and single-bit avalanche of the derived items (t flipped one bit) and of the hash, beside x8. +#[test] +fn stats_beside_x8() { + let key = day_key(DAY); + let cache = Cache::fill_log2(key, 18); + let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 18, derive_len: DERIVE_LEN_X8, state: false }); + let x8 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 18, derive_len: 0, state: false }); + for (label, mp) in [("dr736", &dr), ("x8", &x8)] { + let n = 2048u32; + let mut ones = [0u32; 512]; + let mut flips = 0u64; + let mut flip_n = 0u64; + let mut seen = std::collections::HashSet::new(); + for t in 0..n { + let a = derive_item(t * 2_654_435_761, mp, &cache); + assert!(seen.insert(a), "{label}: duplicate item"); + for i in 0..16 { + for b in 0..32 { + ones[i * 32 + b] += (a[i] >> b) & 1; + } + } + let bit = t % 32; + let c = derive_item((t * 2_654_435_761) ^ (1 << bit), mp, &cache); + for i in 0..16 { + flips += (a[i] ^ c[i]).count_ones() as u64; + } + flip_n += 512; + } + let avalanche = flips as f64 / flip_n as f64 * 100.0; + let worst_z = ones.iter().map(|&o| ((o as f64 - n as f64 / 2.0) / (n as f64 / 4.0).sqrt()).abs()).fold(0.0, f64::max); + println!("{label}: avalanche {avalanche:.2} percent, worst bit z {worst_z:.2}"); + assert!((48.5..=51.5).contains(&avalanche), "{label}: avalanche {avalanche}"); + assert!(worst_z < 4.5, "{label}: worst bit z {worst_z}"); + } + // the hash on the same seeds: avalanche across a nonce flip within the unit + let e = Epoch::new_class("igneum-genesis", DAY, DatasetMode::MemoryHard, 20, LoadClass::DR736); + let w0 = e.hash_warp(0); + let w1 = e.hash_warp(32); + let mut d = 0u64; + for l in 0..32 { + d += (w0[l] ^ w1[l]).count_ones() as u64; + } + let av = d as f64 / (32.0 * 64.0) * 100.0; + println!("hash unit 0 against unit 1: {av:.2} percent of bits differ"); + assert!((44.0..=56.0).contains(&av)); +} + +/// The kernels' text forms: every form's C text, read back into the scalar reference's arithmetic by hand. +#[test] +fn text_forms_match_scalar_reference() { + let mut rng = SplitMix64::new(42); + let p = DeriveProgram::draw_candidate(&mut rng, 64, 0); + let mut s = [0u32; 16]; + for (i, v) in s.iter_mut().enumerate() { + *v = 0x9e37_79b9u32.wrapping_mul(i as u32 + 1); + } + let mut seen = [false; 12]; + for ins in p.rounds.iter().flatten() { + seen[ins.op as usize] = true; + let before = s; + run_round_scalar(std::slice::from_ref(ins), &mut s); + let (d, c, b) = (ins.dst as usize, ins.src as usize, ins.src2 as usize); + let (dv, cv, bv) = (before[d], before[c], before[b]); + let want = match ins.op { + DOp::Add => dv.wrapping_add(cv), + DOp::Sub => dv.wrapping_sub(cv), + DOp::Xor => dv ^ cv, + DOp::Mul => dv.wrapping_mul(cv | 1), + DOp::Rot => dv.rotate_left(ins.rot as u32).wrapping_add(cv), + DOp::XRot => (dv ^ cv).rotate_left(ins.rot as u32), + DOp::AddC => dv.wrapping_add(cv.wrapping_add(ins.imm)), + DOp::XorC => dv ^ cv ^ ins.imm, + DOp::MulC => (dv ^ cv).wrapping_mul(ins.imm), + DOp::MulC2 => dv.wrapping_mul(ins.imm).wrapping_add(cv), + DOp::AndX => dv ^ (cv & bv), + DOp::OrX => dv.wrapping_add(cv | bv), + }; + assert_eq!(s[d], want, "{}", instr_text(ins)); + for i in 0..16 { + if i != d { + assert_eq!(s[i], before[i], "only the destination changes: {}", instr_text(ins)); + } + } + assert!(instr_text(ins).starts_with(&format!("s[{d}]"))); + } + assert!(seen.iter().all(|s| *s), "64 x 9 draws cover every form"); +} + +/// The dataset source of the class on a day: the verifier's `word` path derives through the program. +#[test] +fn dataset_source_word_path() { + let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false }); + let m = ds.memhard().unwrap(); + let item = derive_item(3, &m.params, &m.cache); + for j in 0..16u32 { + assert_eq!(ds.word(3 * 16 + j), item[j as usize]); + } + assert_eq!(ds.word((1 << 20) + 5), ds.word(5), "the mask"); +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/tests/mixer.rs b/tools/attack/adv-accept-v5/igneum-pow/tests/mixer.rs new file mode 100644 index 000000000..f4644ae8d --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/tests/mixer.rs @@ -0,0 +1,356 @@ +//! The class v3 dataset construction (mixer x4, cache growth option C; `docs/plans/mixer-x4.md`): the soundness +//! runs the brief asks for, on the CPU, with the packs for the GPU runs written on request. +//! +//! 1. Fuzz: `IGNEUM_MIXER_FUZZ` (default 200) programs through the seam (`ProgramClass::V3`), the contract on every +//! instruction (the v2 program of the seed, instruction for instruction), 4 units each across the 32-bit range +//! including the wrap, interpreted twice on the CPU; with `IGNEUM_MIXER_PACKS_OUT=` every program is written +//! as a pack with its 4 bases in vectors.json for `packbench` and the OpenCL host (the Metal fuzz). +//! 2. Stats: bit balance and single-bit-flip avalanche of the v3 hash against v2 on the same programs and nonces. +//! 3. Edge: the dataset at word 0, word MASK and the item boundary, derived through the interpreter's fetch path and +//! by hand at every multiplier 1, 2, 4, 8, on a small cache. +//! 4. Determinism: two independent epochs of the same seed and day agree on every vector and every emitted file. +//! +//! Counter ASIC 3.0 gate run (6 October 2026): `IGNEUM_MIXER_CLASS=` (a `LoadClass::parse` name, for example +//! `mx8+sh256x27`) runs the fuzz, the stats and the determinism test on that class instead of `V3_CLASS`; +//! `IGNEUM_MIXER_ERA=igneum-era-test/` composes that era over the class as the chain composes it inside class v3 +//! (`LoadClass::era(class, E_n, &V3_ALLOWED)`, generator 3). The fuzz contract on a shadow class also checks the block: +//! `S` instructions, none a load, every field in range, and the 64 base instructions equal to the class's without the +//! shadow, draw for draw. The edge test is the dataset's alone (the shadow touches no dataset word) and takes no class. + +use igneum_pow::emit::{export_pack, vectors_json}; +use igneum_pow::generator::{era_generator_of, generate_era, generate_from_seed_bytes, generate_from_seed_bytes_class, generate_from_seed_bytes_program_class, EraParams, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, INSTR_COUNT, V3_ALLOWED, V3_CLASS, V4_CLASS, V4_SHADOW_INSTRS}; +use igneum_pow::memhard::{derive_item, mixer, round_key, Cache, MixParams, Shape}; +use igneum_pow::seed::{day_key, SplitMix64}; +use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch}; +use std::collections::HashMap; +use std::path::PathBuf; + +const DAY: &str = "2026-10-03"; + +/// The class under test (`IGNEUM_MIXER_CLASS`, default `V3_CLASS`) and the era composed over it (`IGNEUM_MIXER_ERA`, +/// `igneum-era-test/`, default none). With an era the class returned is the era class (the name carries `-era`). +fn class_under_test() -> (LoadClass, Option<[u8; 32]>) { + let class = std::env::var("IGNEUM_MIXER_CLASS").ok().map(|s| LoadClass::parse(&s).expect("a load class")).unwrap_or(V3_CLASS); + let era = std::env::var("IGNEUM_MIXER_ERA").ok().map(|s| { + assert!(s.starts_with("igneum-era-test/"), "IGNEUM_MIXER_ERA is igneum-era-test/"); + EraParams::test_era_bytes(&s) + }); + match era { + Some(eb) => (LoadClass::era(class, &eb, &V3_ALLOWED), Some(eb)), + None => (class, None), + } +} + +/// The program of `seed` under `class` (an era class carries its era bytes): the seam for `V3_CLASS`, `generate_era` +/// for an era class, the plain class generator otherwise. +fn program_of(seed: &str, class: LoadClass, era: Option<[u8; 32]>) -> Program { + match era { + Some(eb) => generate_era(seed, seed.as_bytes(), LoadClass { era: None, ..class }, &eb, &V3_ALLOWED), + None if class == V3_CLASS => generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V3, None), + None => generate_from_seed_bytes_class(seed, seed.as_bytes(), class), + } +} + +fn contract(p: &Program, seed: &str, class: LoadClass, era: Option<[u8; 32]>) { + if class == V3_CLASS { + assert_eq!(p.generator, GENERATOR_VERSION_V3); + } else if era.is_some() { + // the generator of an era class is the class's (ca3-v4-node 7c22d0d): 4 on V4_CLASS, 3 otherwise + assert_eq!(p.generator, era_generator_of(&LoadClass { era: None, ..class })); + } + assert_eq!(p.class, class); + assert_eq!(p.instrs.len(), INSTR_COUNT); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16); + for (k, i) in p.instrs.iter().enumerate() { + assert!(i.src != i.dst, "#{k}: src == dst"); + assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot); + assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask); + assert!(i.dst < 8 && i.src < 8 && i.src2 < 8); + assert_eq!(i.width, 1, "#{k}: a v3 load reads one word"); + } + assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program"); + if era.is_none() { + // no era: the base program is the v2 program of the seed (the mixer and the shadow draw nothing before it) + let v2 = generate_from_seed_bytes(seed, seed.as_bytes()); + assert_eq!(p.instrs, v2.instrs, "the v2 program of the seed under the v3 construction"); + assert_eq!(p.attempt, v2.attempt); + } + match class.shadow { + None => assert!(p.shadow.is_empty() && !p.has_shadow()), + Some(sh) => { + // the latency-shadow block (Counter ASIC 3.0 item 8): S ALU instructions, no load, every field in range, + // and the base program is the class's without the shadow, draw for draw (the block is drawn after it) + assert_eq!(p.shadow.len(), sh.instrs as usize, "{seed}: shadow block length"); + assert!(p.has_shadow()); + assert_eq!(p.shadow_instrs_per_hash(), sh.instrs as usize * sh.reps as usize * 8); + for (k, i) in p.shadow.iter().enumerate() { + assert!(!i.op.is_load() && i.op != Op::WLoad, "shadow #{k}: a load"); + assert!(i.src != i.dst, "shadow #{k}: src == dst"); + assert!((1..=31).contains(&i.rot), "shadow #{k}: rot {}", i.rot); + assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "shadow #{k}: mask {}", i.mask); + assert!(i.dst < 8 && i.src < 8 && i.src2 < 8 && i.bit < 32, "shadow #{k}: register or bit out of range"); + assert_eq!((i.width, i.win, i.off), (1, 0, 0), "shadow #{k}: no load fields"); + } + let base = program_of(seed, LoadClass { shadow: None, ..class }, era); + let v4_shape = LoadClass { era: None, shadow: None, ..class } == LoadClass { shadow: None, ..V4_CLASS } && sh.instrs == V4_SHADOW_INSTRS; + if v4_shape { + // the amended class v4 (AP-F8-1, sub-version 2): on every draw path a load's source is a register + // fresh by dataflow (a load keeps freshness only from a fresh source; add, sub, xor, mad, shfl from + // either operand; rotates from their operand; or, mul, mulhi never), so its base program is its own + // stream, not class v3's; what holds is the rule itself, checked here on every load site in draw order + let mut fresh = [true; 8]; + for (k, i) in p.instrs.iter().enumerate() { + let (d, a) = (i.dst as usize, i.src as usize); + if i.op.is_load() { + assert!(fresh[a], "{seed}: load #{k} reads r{} which is not fresh by dataflow", i.src); + } + fresh[d] = match i.op { + Op::Load | Op::WLoad | Op::Scratch | Op::Hot => fresh[a], + Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl => fresh[d] || fresh[a], + Op::Rotl | Op::Rotr => fresh[d], + Op::Or | Op::Mul | Op::MulHi => false, + }; + } + } else { + assert_eq!(p.instrs, base.instrs, "{seed}: the base program is the class's without the shadow"); + assert_eq!(p.attempt, base.attempt); + } + if era.is_none() || p.generator == GENERATOR_VERSION_V4 { + // a generator-2 program carries the shadow in its id bytes; a class v4 program is generator 4 + // (ca3-v4-node 7c22d0d), so its id differs from the generator-3 id of the same seeds + assert_ne!(p.program_id(), base.program_id(), "{seed}: the shadow is in the program id"); + } else { + // a generator-3 program's id is program_id(3, seed, attempt), class-independent by construction + // (generator.rs program_id): under the chain's path a class v4 program shares its id with the class + // v3 program of the same seeds until the v4 seam gives it its own generator or puts the class in the + // id (Counter ASIC 3.0 gate run, 6 October 2026: an item for gates G4 and G6, not a hash fault) + assert_eq!(p.generator, GENERATOR_VERSION_V3); + if p.program_id() == base.program_id() { + println!("{seed}: generator-3 program id {:016x} is the same with and without the shadow (the v4 seam item)", p.program_id()); + } + } + } + } +} + +/// Write a pack whose vectors.json carries `bases` instead of the three standard bases (packbench and the OpenCL +/// host check every unit standalone and the ones inside the batch window). +fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) { + let mut pack = export_pack(e, day, source); + let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect(); + let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true); + for f in pack.files.iter_mut() { + if f.0 == "vectors.json" { + f.1 = vj.clone(); + } + } + pack.write_to(dir).unwrap(); +} + +#[test] +fn fuzz_v3_programs_cpu() { + let n: usize = std::env::var("IGNEUM_MIXER_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200); + // IGNEUM_FUZZ_SEED_BASE (default 0) offsets the seed index so a continuous fuzzer (the box's capacity layer, + // infra/build-server/capacity) walks fresh programs round after round; the default run is unchanged + let base: usize = std::env::var("IGNEUM_FUZZ_SEED_BASE").ok().and_then(|s| s.parse().ok()).unwrap_or(0); + let out = std::env::var("IGNEUM_MIXER_PACKS_OUT").ok().map(PathBuf::from); + // IGNEUM_MIXER_CLASS=mx8 fuzzes the x8 candidate as a load class (generator 2 with the class in the id); the + // default is V3_CLASS through the seam; IGNEUM_MIXER_ERA composes a test era over the class (class_under_test) + let (class, era) = class_under_test(); + let mut rng = SplitMix64::new(0x6967_6e65_756d_2d6d); // "igneum-m" + let shape = Shape::for_class(&class); + assert_eq!(shape.cache_log2_words, 26); + assert!(shape.mixer_mult > 1); + // one memory-hard source per dataset size (the 256 MiB cache fill is 0.2 s each) + let mut mh: HashMap = HashMap::new(); + let mut manifest = String::from("pack\tlog2\tprogram_id\tbases\n"); + let mut units = 0usize; + let mut wraps = 0usize; + for i in base..base + n { + let seed = format!("igneum-mixer-fuzz/{i}"); + let p = program_of(&seed, class, era); + contract(&p, &seed, class, era); + let b0 = (rng.below(8) as u32) * 32; + let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32); + let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32); + let b3 = (rng.next() as u32) & !31; + let bases = [b0, b1, b2, b3]; + wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count(); + let log2 = [24u32, 26, 28][rng.below(3) as usize]; + let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, log2, shape)); + let e = Epoch { program: p, dataset: ds }; + for &b in &bases { + let r1 = e.interpret_warp(b); + let r2 = e.interpret_warp(b); + assert_eq!(r1.hashes, r2.hashes); + assert!(r1.items_derived >= 120 * 32 / 32 && r1.items_derived <= 4_096, "{seed}: {} items", r1.items_derived); + units += 1; + } + if let Some(dir) = &out { + let pack_name = format!("fuzz-{i:03}-{}-l{log2}", class.name()); + write_pack_with_bases(&dir.join(&pack_name), &e, DAY, &bases, "igneum-pow tests/mixer.rs fuzz"); + manifest.push_str(&format!( + "{pack_name}\t{log2}\t{:016x}\t{}\n", + e.program.program_id(), + bases.iter().map(|b| format!("{b}")).collect::>().join(",") + )); + } + mh.insert(log2, e.dataset); + } + println!("fuzz: {n} {} programs, {units} units on the CPU, {wraps} units in the top 256 nonces, seeds {base}..{}", class.name(), base + n); + assert_eq!(units, 4 * n); + assert_eq!(wraps, n); + if let Some(dir) = &out { + std::fs::create_dir_all(dir).unwrap(); + std::fs::write(dir.join("manifest.tsv"), manifest).unwrap(); + println!("packs written to {}", dir.display()); + } +} + +/// Bit balance and avalanche of the v3 hash beside v2 on the same program (the TESTS.md section 3 shape, on the +/// CPU, 2^13 nonces per seed): every output bit within 5 sigma of half ones; a single nonce-bit flip moves 50 percent +/// of the output bits within 2 points; no duplicate among the outputs. +#[test] +fn stats_v3_against_v2() { + let n_warps = 256usize; // 8,192 nonces + let (class, era) = class_under_test(); + let class_name = if class == V3_CLASS { "v3".to_string() } else { class.name() }; + for seed in ["igneum-genesis", "igneum-genesis/stats1"] { + let v3 = Epoch { + program: program_of(seed, class, era), + dataset: DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 24, Shape::for_class(&class)), + }; + let v2 = Epoch::new(seed, DAY, DatasetMode::MemoryHard, 24); + for (name, e) in [(class_name.as_str(), &v3), ("v2", &v2)] { + let mut ones = [0u64; 64]; + let mut outs = Vec::with_capacity(n_warps * 32); + for w in 0..n_warps { + let h = e.hash_warp(w as u32 * 32); + for &x in &h { + outs.push(x); + for b in 0..64 { + ones[b] += (x >> b) & 1; + } + } + } + let total = (n_warps * 32) as f64; + let sigma = (total / 4.0).sqrt(); + for (b, &c) in ones.iter().enumerate() { + let z = (c as f64 - total / 2.0).abs() / sigma; + assert!(z < 5.0, "{seed} {name}: bit {b} ones {c} of {total}, z {z:.2}"); + } + // avalanche: flip one bit of the nonce within the unit (lanes 0..31 differ in the low 5 bits) and across + // units (bit 5 and up): compare lane l of unit u with lane l ^ (1 << k) and with unit u ^ (1 << k) + let mut flips = 0u64; + let mut moved = 0u64; + for w in 0..64usize { + let h = e.hash_warp(w as u32 * 32); + for k in 0..5 { + for l in 0..32usize { + moved += (h[l] ^ h[l ^ (1 << k)]).count_ones() as u64; + flips += 1; + } + } + let h2 = e.hash_warp((w ^ 1) as u32 * 32); + for l in 0..32usize { + moved += (h[l] ^ h2[l]).count_ones() as u64; + flips += 1; + } + } + let avg = moved as f64 / flips as f64 / 64.0 * 100.0; + assert!((avg - 50.0).abs() < 2.0, "{seed} {name}: avalanche {avg:.2} percent"); + outs.sort_unstable(); + let dups = outs.windows(2).filter(|p| p[0] == p[1]).count(); + assert_eq!(dups, 0, "{seed} {name}: duplicate outputs"); + println!("{seed} {name}: {} outputs, avalanche {avg:.2} percent, worst bit z {:.2}", outs.len(), ones.iter().map(|&c| (c as f64 - total / 2.0).abs() / sigma).fold(0.0, f64::max)); + } + assert_ne!(v3.hash_warp(0), v2.hash_warp(0)); + } +} + +/// The dataset edges under every multiplier on a small cache: word 0, word MASK, the last word of item 0 and the +/// first of item 1, through `DatasetSource::word` and by hand. +#[test] +fn edge_items_every_multiplier() { + let key = day_key(DAY); + let cache = Cache::fill_log2(key, 14); + for m in [1u32, 2, 4, 8] { + let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 14, derive_len: 0, state: false }); + let by_hand = |t: u32| -> [u32; 16] { + let mut s = [0u32; 16]; + s[..8].copy_from_slice(&key); + for i in 0..8 { + s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); + } + for r in 0..8usize { + for j in 0..m as usize { + mixer(&mut s, round_key(r * m as usize + j), &mp); + } + let line = cache.line(s[0]); + for i in 0..16 { + s[i] ^= line[i]; + } + } + for j in 0..m as usize { + mixer(&mut s, round_key(8 * m as usize + j), &mp); + } + s + }; + for t in [0u32, 1, 0x0fff_ffff, 0xffff_ffff] { + assert_eq!(derive_item(t, &mp, &cache), by_hand(t), "m {m} item {t}"); + } + } + // the interpreter's fetch path at the genesis cache: words 0, 15, 16 and MASK of a 2^20-word dataset agree with + // the item derivation, under v3 + let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape::for_class(&V3_CLASS)); + let m = ds.memhard().unwrap(); + for w in [0u32, 15, 16, 17, ds.mask - 1, ds.mask] { + assert_eq!(ds.word(w), derive_item(w >> 4, &m.params, &m.cache)[(w & 15) as usize]); + assert_eq!(ds.word(w), m.word(w)); + } + // a load at an out-of-range register masks to the dataset: the word at mask + 1 is the word at 0 + assert_eq!(ds.word(ds.mask.wrapping_add(1)), ds.word(0)); +} + +/// Two independent epochs of the same seed and day: every vector and every emitted file identical; the pinned v3 +/// pack is what a third export writes. +#[test] +fn determinism_v3() { + let (class, era) = class_under_test(); + let build = || Epoch { + program: program_of("igneum-genesis", class, era), + dataset: DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 28, Shape::for_class(&class)), + }; + let a = build(); + let b = build(); + let pa = export_pack(&a, DAY, "a"); + let pb = export_pack(&b, DAY, "a"); + assert_eq!(pa.outs, pb.outs); + assert_eq!(pa.vectors, pb.vectors); + assert_eq!(pa.files, pb.files); + for (w, warp) in [(0u32, 0usize), (4096, 1), (1_000_000, 2)] { + assert_eq!(a.hash_warp(w), pa.outs[warp]); + } + // the pinned pack of the class: mx8-genesis for V3_CLASS, packs-ca3-shadow/ for a shadow class over mx8 + // (igneum-genesis, the same day); a class with no pinned pack skips the on-disk comparison and says so + let pinned = if class == V3_CLASS { + Some("../proto-cuda/packs-ca2-mixer/mx8-genesis".to_string()) + } else { + class.name().strip_prefix("mx8+").map(|b| format!("../proto-cuda/packs-ca3-shadow/{b}")) + }; + let dir = match pinned.map(|d| PathBuf::from(env!("CARGO_MANIFEST_DIR")).join(d)).filter(|d| d.is_dir()) { + Some(d) => d, + None => { + println!("determinism {}: two builds equal; no pinned pack on disk for this class", class.name()); + return; + } + }; + println!("determinism {}: two builds equal, against the pinned pack {}", class.name(), dir.display()); + for (name, text) in &pa.files { + if name == "vectors.json" || name == "vectors.h" { + continue; // the source string differs ("a" here) + } + let on_disk = std::fs::read_to_string(dir.join(name)).unwrap(); + assert_eq!(&on_disk, text, "{name}"); + } +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/tests/packs.rs b/tools/attack/adv-accept-v5/igneum-pow/tests/packs.rs new file mode 100644 index 000000000..f66d91536 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/tests/packs.rs @@ -0,0 +1,999 @@ +//! The checked-in packs under proto-cuda/packs/ against this crate (their source since generator version 2, +//! 4 October 2026): every program instruction by instruction from its seed bytes, the generator version, attempt +//! and program id, the dataset and cache self-test words, the 96 hash vectors per pack, and every emitted file +//! byte for byte. A pack that drifts from the emitters, or a generator change that moves a vector, fails here. +//! +//! Packs: igneum-genesis-mh and igneum-devnet-v4-epoch0 (memory-hard; the latter from the devnet genesis hash as +//! the epoch seed and the day bytes of 2026-10-04), igneum-genesis and igneum-hourly (closed-form dataset, +//! interpreter regression only); and, under `proto-cuda/packs-ca2-mixer/`, the class v3 packs mx8-genesis and +//! mx8-devnet-epoch0 (Counter ASIC 2.0, 5 October 2026: generator 3 on `V3_CLASS` = mixer x8 with the cache growth +//! rule, decided 22:05 UTC under the delegated rule; the same seeds and days as the two memory-hard v2 packs, so the +//! v2 program and cache carry over and only the dataset words and the hashes change) and the x4 candidate's packs +//! mx4-genesis and mx4-devnet-epoch0 (generator 2 with the load class in the id, the record of the x4 rows). + +use igneum_pow::accept; +use igneum_pow::emit::{ + cuda_kernel, cuda_kernel_bound, cuda_memhard_header, export_pack, metal_memhard, metal_memhard_for, metal_program, + metal_program_bound, opencl_kernel, opencl_kernel_bound, program_header, program_json, LoadSource, +}; +use igneum_pow::generator::{generate_from_seed_bytes, generate_from_seed_bytes_class, generate_from_seed_bytes_program_class, LoadClass, Op, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, LOAD_SLOTS, V3_CLASS}; +use igneum_pow::memhard::{Shape, CACHE_WORDS}; +use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch}; +use serde_json::Value; +use std::path::PathBuf; +use std::sync::OnceLock; + +const PACKS: [&str; 8] = [ + "igneum-genesis-mh", + "igneum-devnet-v4-epoch0", + "igneum-genesis", + "igneum-hourly", + "mx8-genesis", + "mx8-devnet-epoch0", + "mx4-genesis", + "mx4-devnet-epoch0", +]; +/// The class v3 packs (generator 3 through the seam). +const PACKS_V3: [&str; 2] = ["mx8-genesis", "mx8-devnet-epoch0"]; + +fn packs_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs") +} + +/// The directory a pack lives in: the class v3 packs under packs-ca2-mixer, the rest under packs. +fn pack_dir(pack: &str) -> PathBuf { + if pack.starts_with("mx4-") || pack.starts_with("mx8-") { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-mixer").join(pack) + } else { + packs_dir().join(pack) + } +} + +fn read(pack: &str, file: &str) -> String { + let p = pack_dir(pack).join(file); + std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display())) +} + +fn json(pack: &str, file: &str) -> Value { + serde_json::from_str(&read(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")) +} + +fn hex32(v: &Value) -> u32 { + u32::from_str_radix(v.as_str().unwrap().trim_start_matches("0x"), 16).unwrap() +} +fn hex64(v: &Value) -> u64 { + u64::from_str_radix(v.as_str().unwrap().trim_start_matches("0x"), 16).unwrap() +} +fn unhex(v: &Value) -> Vec { + igneum_pow::bind::unhex(v.as_str().unwrap()).unwrap() +} + +/// The epoch a pack describes, rebuilt from program.json alone: the program from `seed_bytes`, the dataset from +/// `dataset.day_bytes` in the pack's mode and size. Memory-hard packs fill a 256 MiB cache (about 0.2 s), so +/// each is built once. +fn epoch(pack: &str) -> &'static Epoch { + static E: OnceLock> = OnceLock::new(); + let all = E.get_or_init(|| { + PACKS + .iter() + .map(|p| { + let j = json(p, "program.json"); + let seed = j["seed"].as_str().unwrap(); + let seed_bytes = unhex(&j["seed_bytes"]); + let day_bytes = unhex(&j["dataset"]["day_bytes"]); + let mode = match j["dataset_mode"].as_str().unwrap() { + "memory-hard" => DatasetMode::MemoryHard, + _ => DatasetMode::ClosedForm, + }; + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + // a class v3 pack: generator 3 on V3_CLASS through the seam, the era bytes it records, the dataset + // in the class's shape on day 0 (the growth rule's genesis cache: every pinned pack is a day-0 size) + let program = match j["generator"].as_u64().unwrap() as u32 { + GENERATOR_VERSION_V3 => { + let era = j.get("era_seed_bytes").map(unhex); + generate_from_seed_bytes_program_class(seed, &seed_bytes, ProgramClass::V3, era.as_deref()) + } + // a generator 2 pack with a load class (the x4 candidate's packs): the class from program.json + _ => match j.get("load_class").and_then(|c| c.as_str()) { + Some(c) => generate_from_seed_bytes_class(seed, &seed_bytes, LoadClass::parse(c).expect("a load class name")), + None => generate_from_seed_bytes(seed, &seed_bytes), + }, + }; + let shape = Shape::for_class(&program.class); + let mut dataset = + DatasetSource::from_key_shape(igneum_pow::seed::seed_words_from_bytes(&day_bytes), mode, log2, shape); + dataset.key_bytes = day_bytes; + (p.to_string(), Epoch { program, dataset }) + }) + .collect() + }); + &all.iter().find(|(n, _)| n == pack).unwrap().1 +} + +fn day_label(pack: &str) -> String { + json(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string() +} + +fn check_program_json(pack: &str) { + let j = json(pack, "program.json"); + let p = &epoch(pack).program; + assert_eq!(j["format"].as_str().unwrap(), "igneum-program-pack-3"); + assert_eq!(j["generator"].as_u64().unwrap() as u32, p.generator, "{pack}: generator version"); + assert_eq!(p.generator, if PACKS_V3.contains(&pack) { GENERATOR_VERSION_V3 } else { GENERATOR_VERSION }); + assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt, "{pack}: attempt"); + assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id"); + let sw: Vec = j["seed_words"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(p.seed.to_vec(), sw, "{pack}: seed words"); + assert_eq!(p.loads_per_hash() as u64, j["loads_per_hash"].as_u64().unwrap(), "{pack}: loads per hash"); + assert_eq!(p.loads_per_hash(), 8 * LOAD_SLOTS); + assert!(accept::check(p).is_ok(), "{pack}: the pack's program passes the acceptance rule"); + let instrs = j["instructions"].as_array().unwrap(); + assert_eq!(instrs.len(), p.instrs.len(), "{pack}: instruction count"); + for (k, (ins, ji)) in p.instrs.iter().zip(instrs).enumerate() { + assert_eq!(ji["i"].as_u64().unwrap() as usize, k); + assert_eq!(Op::from_name(ji["op"].as_str().unwrap()).unwrap(), ins.op, "{pack} #{k} op"); + assert_eq!(ji["dst"].as_u64().unwrap(), ins.dst as u64, "{pack} #{k} dst"); + assert_eq!(ji["src"].as_u64().unwrap(), ins.src as u64, "{pack} #{k} src"); + assert_eq!(ji["src2"].as_u64().unwrap(), ins.src2 as u64, "{pack} #{k} src2"); + assert_eq!(hex32(&ji["imm"]), ins.imm, "{pack} #{k} imm"); + assert_eq!(hex32(&ji["imm2"]), ins.imm2, "{pack} #{k} imm2"); + assert_eq!(ji["rot"].as_u64().unwrap(), ins.rot as u64, "{pack} #{k} rot"); + assert_eq!(ji["bit"].as_u64().unwrap(), ins.bit as u64, "{pack} #{k} bit"); + assert_eq!(ji["mask"].as_u64().unwrap(), ins.mask as u64, "{pack} #{k} mask"); + } + let mix = j["op_mix"].as_object().unwrap(); + for (name, count) in p.histogram() { + assert_eq!(mix[name].as_u64().unwrap() as usize, count, "{pack}: op_mix {name}"); + } + assert_eq!(mix.len(), p.histogram().len()); +} + +#[test] +fn program_json_matches_all_packs() { + for pack in PACKS { + check_program_json(pack); + } +} + +/// The genesis program of generator v2 (spec 01 section 1.4.3 vector). +#[test] +fn genesis_program_shape() { + let p = &epoch("igneum-genesis-mh").program; + assert_eq!(p.attempt, 0); + assert_eq!(p.op_mix(), "load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1"); + assert_eq!(p.program_id(), 0xbcc1248b10cc90f2); + assert_eq!(&epoch("igneum-genesis").program, p, "closed-form and memory-hard packs share the program"); +} + +#[test] +fn mixer_params_match_pack() { + for pack in ["igneum-genesis-mh", "igneum-devnet-v4-epoch0", "mx8-genesis", "mx8-devnet-epoch0", "mx4-genesis", "mx4-devnet-epoch0"] { + let j = json(pack, "program.json"); + let mp = &epoch(pack).dataset.memhard().unwrap().params; + let key: Vec = j["dataset"]["key"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(mp.key.to_vec(), key); + let rot: Vec = + j["dataset"]["mixer"]["rot"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u32).collect(); + assert_eq!(mp.rot.to_vec(), rot); + let mul: Vec = j["dataset"]["mixer"]["mul"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(mp.mul.to_vec(), mul); + let rc: Vec = j["dataset"]["mixer"]["rc"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(mp.rc.to_vec(), rc); + assert_eq!(hex32(&j["dataset"]["d0"]), mp.key[0]); + assert_eq!(hex32(&j["dataset"]["d1"]), mp.key[1]); + } +} + +#[test] +fn cache_matches_vectors() { + let v = json("igneum-genesis-mh", "vectors.json"); + let m = epoch("igneum-genesis-mh").dataset.memhard().unwrap(); + let w = m.cache.words(); + assert_eq!(w.len(), CACHE_WORDS); + let head: Vec = v["cache_head"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(&w[..16], &head[..]); + let last: Vec = v["cache_last_line"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(&w[CACHE_WORDS - 16..], &last[..]); + assert_eq!(m.cache.fnv1a64(), 0x48c4f5bf24166b2e, "cache FNV-1a 64 of day 2026-10-03 (MEMHARD.md), unchanged by v2"); + assert_eq!(m.cache.fnv1a64(), hex64(&v["cache_fnv1a64"])); + let v = json("igneum-devnet-v4-epoch0", "vectors.json"); + let m = epoch("igneum-devnet-v4-epoch0").dataset.memhard().unwrap(); + assert_eq!(m.cache.fnv1a64(), hex64(&v["cache_fnv1a64"])); + assert_eq!(m.cache.fnv1a64(), 0x448274a57f508cbc, "cache FNV-1a 64 of day bytes igneum-day/20730"); +} + +fn check_dataset_words(pack: &str) { + let v = json(pack, "vectors.json"); + let e = epoch(pack); + let ds = &e.dataset; + // a pack's self-test words are read under its program's layout (the era layout; linear for every v2 pack) + let head: Vec = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect(); + for (i, h) in head.iter().enumerate() { + assert_eq!(e.dataset_word(i as u32), *h, "{pack}: dataset[{i}]"); + } + let last_index = v["dataset_last_index"].as_u64().unwrap() as u32; + assert_eq!(last_index, ds.mask); + assert_eq!(e.dataset_word(last_index), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]"); + let samples = v["dataset_samples"].as_array().unwrap(); + assert_eq!(samples.len(), 64); + for s in samples { + let idx = s["index"].as_u64().unwrap() as u32; + assert_eq!(e.dataset_word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]"); + } +} + +#[test] +fn dataset_words_match_all_packs() { + for pack in PACKS { + check_dataset_words(pack); + } +} + +/// Returns the number of hashes compared (3 warps x 32 lanes = 96). +fn check_vectors(pack: &str) -> usize { + let v = json(pack, "vectors.json"); + let e = epoch(pack); + assert_eq!(v["dataset_mode"].as_str().unwrap(), e.dataset.mode().name()); + assert_eq!(v["dataset_log2_words"].as_u64().unwrap() as u32, e.dataset.log2_words); + let warps = v["warps"].as_array().unwrap(); + assert_eq!(warps.len(), 3); + let mut n = 0; + for w in warps { + let base = w["base_nonce"].as_u64().unwrap() as u32; + let expected: Vec = w["expected"].as_array().unwrap().iter().map(hex64).collect(); + let got = e.hash_warp(base); + for lane in 0..32 { + assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}"); + n += 1; + } + // The single-nonce API agrees with the warp. + assert_eq!(e.hash(base + 7), expected[7]); + assert!(e.verify_block(base + 7, expected[7])); + assert!(!e.verify_block(base + 7, expected[7] - 1)); + } + n +} + +#[test] +fn vectors_96_per_pack() { + for pack in PACKS { + assert_eq!(check_vectors(pack), 96, "{pack}"); + } +} + +/// The spec 01 section 1.17 vectors (igneum-genesis-mh, generator v2). +#[test] +fn spec_vectors_genesis_mh() { + let e = epoch("igneum-genesis-mh"); + let w = e.hash_warp(0); + assert_eq!(w[0], 0x42246ba99fc58e4f); + assert_eq!(w[31], 0xb08446b1f2de7793); + assert_eq!(e.hash_warp(4096)[0], 0x3d3903e310ca038f); + assert_eq!(e.hash_warp(1_000_000)[0], 0xf218c1bd58e6dfe0); +} + +fn assert_same_text(pack: &str, file: &str, got: &str) { + let want = read(pack, file); + if got != want { + let (gl, wl): (Vec<&str>, Vec<&str>) = (got.lines().collect(), want.lines().collect()); + for i in 0..gl.len().max(wl.len()) { + let g = gl.get(i).copied().unwrap_or(""); + let w = wl.get(i).copied().unwrap_or(""); + if g != w { + panic!("{pack}/{file} differs at line {}:\n pack: {w}\n rust: {g}", i + 1); + } + } + panic!("{pack}/{file} differs only in trailing bytes (len {} vs {})", got.len(), want.len()); + } +} + +fn check_sources(pack: &str) { + let e = epoch(pack); + let p = &e.program; + let day = day_label(pack); + let mp = e.dataset.memhard().map(|m| &m.params); + assert_same_text(pack, "kernel.cu", &cuda_kernel(p, mp)); + assert_same_text(pack, "kernel_bound.cu", &cuda_kernel_bound(p, mp)); + assert_same_text(pack, "program.metal", &metal_program(p, e.dataset.log2_words, LoadSource::Stored)); + assert_same_text(pack, "program_bound.metal", &metal_program_bound(p, e.dataset.log2_words)); + assert_same_text(pack, "kernel.cl", &opencl_kernel(p, mp)); + assert_same_text(pack, "kernel_bound.cl", &opencl_kernel_bound(p, mp)); + assert_same_text(pack, "program.h", &program_header(p, &day, &e.dataset)); + if let Some(mp) = mp { + assert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp)); + assert_same_text(pack, "memhard.metal", &igneum_pow::emit::metal_memhard_layout(mp, p.class.layout())); + } + let got = program_json(p, &day, &e.dataset); + assert_same_text(pack, "program.json", &got); + let _: Value = serde_json::from_str(&got).expect("program.json is valid JSON"); + // Every load in every emitted hash kernel has the masked form, and there are exactly 16 of them (a class with + // the era layout inside has the era form instead: `((rotl_imm(rN * M, R) & WM) | OFF) & mask`, checked by + // era_emitted_sources_match_and_loads_have_the_era_form over the era packs, and here by the same count). + let era_load = if p.class.era.is_some() { "((rotl_imm(r" } else { "" }; + for (file, load, masked) in [ + ("kernel.cu", "ds[r", " & mask]"), + ("kernel_bound.cu", "ds[r", " & mask]"), + ("program.metal", "dataset[r", " & MASK]"), + ("program_bound.metal", "dataset[r", " & MASK]"), + ] { + let text = read(pack, file); + let load = if era_load.is_empty() { load } else { era_load }; + assert_eq!(text.matches(load).count(), LOAD_SLOTS, "{pack}/{file}: 16 loads"); + assert_eq!(text.matches(masked).count(), LOAD_SLOTS, "{pack}/{file}: 16 masked loads"); + } +} + +#[test] +fn emitted_sources_match_all_packs() { + for pack in PACKS { + check_sources(pack); + } +} + +/// The whole pack as `export` writes it: vectors.json and vectors.h match, and the file list is the full set. +/// The class v3 packs (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): generator 3 on V3_CLASS = mx8; the program of +/// each is the v2 program of the same seed instruction for instruction (v2 loads take no width roll); the cache is +/// the v2 cache (day 0 of the growth rule: 2^26 words, the same FNV-1a 64); the dataset words differ from v2's; +/// program.json, program.h and the emitted memhard core say so; the id carries generator 3. +#[test] +fn v3_packs_are_the_v2_seeds_under_mixer_x8() { + assert_eq!(V3_CLASS.name(), "mx8"); + assert_eq!(V3_CLASS.mixer_mult, 8); + assert!(V3_CLASS.growth); + assert_eq!(LoadClass { mixer_mult: 4, ..V3_CLASS }, LoadClass::MX4, "the x4 candidate differs from v3 in the multiplier alone"); + for (v3, v2) in [("mx8-genesis", "igneum-genesis-mh"), ("mx8-devnet-epoch0", "igneum-devnet-v4-epoch0")] { + let e3 = epoch(v3); + let e2 = epoch(v2); + let j = json(v3, "program.json"); + assert_eq!(j["program_class"].as_str().unwrap(), "v3"); + assert!(j["load_class"].as_str().unwrap().starts_with("mx8"), "{v3}: mx8, or mx8 with the era inside"); + assert_eq!(j["mixer_mult"].as_u64().unwrap(), 8); + assert_eq!(j["cache_growth"].as_bool().unwrap(), true); + assert_eq!(j["dataset"]["mixer_mult"].as_u64().unwrap(), 8); + assert_eq!(j["dataset"]["cache"]["log2_words"].as_u64().unwrap(), 26); + assert_eq!(e3.program.generator, GENERATOR_VERSION_V3); + if e3.program.era_bytes.is_some() { + // the era layout composed into class v3 (docs/plans/era-layout.md, 5 October 2026): a chain pack carries an + // era, so its class is V3_CLASS with the era drawn inside and its stream takes two window draws per + // instruction; the v2 program carries over only in the seed, the attempt and the day + assert_eq!(LoadClass { era: None, ..e3.program.class }, V3_CLASS, "{v3}: the composed class"); + assert!(e3.program.class.era.is_some()); + assert_ne!(e3.program.instrs, e2.program.instrs, "{v3}: the era windows change the stream"); + } else { + assert_eq!(e3.program.class, V3_CLASS); + assert_eq!(e3.program.instrs, e2.program.instrs, "{v3}: the v2 program under the v3 construction"); + } + assert_eq!(e3.program.seed, e2.program.seed); + assert_eq!(e3.program.attempt, e2.program.attempt); + assert_ne!(e3.program.program_id(), e2.program.program_id()); + assert_eq!(e3.program.program_id(), igneum_pow::generator::program_id(GENERATOR_VERSION_V3, &e3.program.seed, e3.program.attempt)); + let m3 = e3.dataset.memhard().unwrap(); + let m2 = e2.dataset.memhard().unwrap(); + assert_eq!(m3.shape(), Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: false }); + assert_eq!(m3.cache.fnv1a64(), m2.cache.fnv1a64(), "{v3}: the same cache as v2 on day 0"); + assert_eq!(m3.params.rot, m2.params.rot); + assert_eq!(e3.dataset.log2_words, 28); + assert_ne!(e3.dataset.word(0), e2.dataset.word(0), "{v3}: the dataset words differ"); + assert_ne!(e3.hash_warp(0), e2.hash_warp(0)); + let h = read(v3, "program.h"); + assert!(h.contains("#define IGNEUM_GENERATOR 3\n")); + assert!(h.contains("#define IGNEUM_PROGRAM_CLASS \"v3\"\n")); + assert!(h.contains("#define IGNEUM_MIXER_MULT 8")); + assert!(h.contains("#define IGNEUM_CACHE_GROWTH 1")); + assert!(h.contains("#define IGNEUM_CACHE_LOG2_WORDS 26\n")); + assert!(h.contains("#define IGNEUM_LOAD_CLASS \"mx8"), "mx8, or mx8 with the era inside"); + for file in ["memhard.h", "memhard.metal", "kernel.cl"] { + let text = read(v3, file); + assert_eq!(text.matches("j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u))").count(), 1, "{v3}/{file}"); + assert_eq!(text.matches("j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u))").count(), 1, "{v3}/{file}"); + } + for file in ["memhard.h", "memhard.metal", "kernel.cl"] { + let text = read(v2, file); + assert_eq!(text.matches("j < 8u").count(), 0, "{v2}/{file}: the v2 text has no multiplier loop"); + } + } + // the devnet v3 pack records era 0's stand-in, the devnet genesis hash + let j = json("mx8-devnet-epoch0", "program.json"); + assert_eq!(unhex(&j["era_seed_bytes"]), unhex(&j["seed_bytes"])); + assert!(read("mx8-devnet-epoch0", "program.h").contains("#define IGNEUM_ERA_SEED_HEX \"edc4fa844da9dc98")); + assert!(json("mx8-genesis", "program.json").get("era_seed_bytes").is_none()); + // the x4 candidate's packs: generator 2, the class in the id, the v2 program of the seed, mixer x4 + for (x4, v2) in [("mx4-genesis", "igneum-genesis-mh"), ("mx4-devnet-epoch0", "igneum-devnet-v4-epoch0")] { + let e4 = epoch(x4); + let j = json(x4, "program.json"); + assert_eq!(e4.program.generator, GENERATOR_VERSION); + assert_eq!(j["load_class"].as_str().unwrap(), "mx4"); + assert_eq!(e4.program.class, LoadClass::MX4); + assert_eq!(e4.program.instrs, epoch(v2).program.instrs); + assert_eq!(e4.dataset.memhard().unwrap().shape(), Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0, state: false }); + assert!(read(x4, "memhard.h").contains("j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u))")); + } +} + +fn check_export(pack: &str) { + let e = epoch(pack); + let v = json(pack, "vectors.json"); + let source = v["source"].as_str().unwrap(); + let out = export_pack(e, &day_label(pack), source); + let file = |name: &str| -> &str { &out.files.iter().find(|(n, _)| n == name).unwrap().1 }; + assert_same_text(pack, "vectors.json", file("vectors.json")); + assert_same_text(pack, "vectors.h", file("vectors.h")); + let mut expected = vec![ + "program.json", + "vectors.json", + "kernel.cu", + "kernel.cl", + "program.h", + "vectors.h", + "program.metal", + "program_bound.metal", + "kernel_bound.cu", + "kernel_bound.cl", + ]; + if e.dataset.mode() == DatasetMode::MemoryHard { + expected.extend(["memhard.h", "memhard.metal"]); + } + assert_eq!(out.files.iter().map(|(n, _)| n.as_str()).collect::>(), expected); + let mut on_disk: Vec = std::fs::read_dir(pack_dir(pack)) + .unwrap() + .map(|d| d.unwrap().file_name().to_string_lossy().to_string()) + .filter(|n| !n.starts_with('.')) + .collect(); + on_disk.sort(); + let mut want: Vec = expected.iter().map(|s| s.to_string()).collect(); + want.sort(); + assert_eq!(on_disk, want, "{pack}: no stale file in the pack directory"); +} + +#[test] +fn export_pack_matches_all_packs() { + for pack in PACKS { + check_export(pack); + } +} + +#[test] +fn item_is_independent_of_dataset_size() { + // MEMHARD.md 1.7: a smaller dataset is a prefix of items, so dataset[w] is the same at every size. + let big = &epoch("igneum-genesis-mh").dataset; + let m = big.memhard().unwrap(); + for w in [0u32, 1, 15, 16, 17, 0x00ff_ffff, 0x03ff_ffff] { + assert_eq!(big.word(w), m.word(w)); + } +} + +/// The devnet pack is the chain's own derivation: epoch seed = the devnet genesis hash, day bytes = `bind::day_bytes(20730)`. +#[test] +fn devnet_pack_is_the_chain_derivation() { + let j = json("igneum-devnet-v4-epoch0", "program.json"); + let genesis = igneum_pow::bind::unhex("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07").unwrap(); + assert_eq!(unhex(&j["seed_bytes"]), genesis); + assert_eq!(unhex(&j["dataset"]["day_bytes"]), igneum_pow::bind::day_bytes(20_730).to_vec()); + let e = Epoch::from_seed_bytes(&genesis, &igneum_pow::bind::day_bytes(20_730), "devnet"); + assert_eq!(e.program.instrs, epoch("igneum-devnet-v4-epoch0").program.instrs); + assert_eq!(e.hash_warp(0), epoch("igneum-devnet-v4-epoch0").hash_warp(0)); +} + +// --------------------------------------------------------------------------------------------------------- +// Era layout packs (5 October 2026, docs/plans/era-layout.md): proto-cuda/packs-ca2-era/era-, n in 0..5, the +// devnet epoch seed and day bytes under the era class of test era seed igneum-era-test/, the width pinned at +// 4 bytes (allowed_widths in program.json). Checked like the pinned packs, plus the one load form of 1.3 by text search. +// --------------------------------------------------------------------------------------------------------- + +use igneum_pow::generator::{generate_era, EraParams, V3_ALLOWED}; + +const ERA_PACKS: [&str; 6] = ["era-0", "era-1", "era-2", "era-3", "era-4", "era-5"]; + +fn era_packs_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-era") +} + +fn era_read(pack: &str, file: &str) -> String { + let p = era_packs_dir().join(pack).join(file); + std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display())) +} + +fn era_json(pack: &str, file: &str) -> Value { + serde_json::from_str(&era_read(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")) +} + +/// The era class of a pack: the era bytes from program.json (`era_seed_bytes`, the chain's `E_n`; the test seed +/// `igneum-era-test/` of the pack's number gives the same bytes), the allowed set from `era.allowed_widths` +/// (class v3's `V3_ALLOWED`); the pack's recorded stream words and draw must be the class's. +fn era_class(pack: &str) -> LoadClass { + let j = era_json(pack, "program.json"); + let n: u64 = pack.trim_start_matches("era-").parse().unwrap(); + let eb = unhex(&j["era_seed_bytes"]); + assert_eq!(eb, EraParams::test_era_bytes(&format!("igneum-era-test/{n}")).to_vec(), "{pack}: the era bytes of test seed {n}"); + let allowed: Vec = j["era"]["allowed_widths"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u8).collect(); + assert_eq!(allowed, V3_ALLOWED.to_vec(), "{pack}: class v3's width set"); + // the measurement packs of 5 October 2026: the era layout over version 2's construction (mixer x1, the genesis + // cache), generator 3 and the era bytes recorded; the chain's class v3 composes the same draw over LoadClass::MX4 + // (generator tests era_programs_are_accepted and program_classes), and the integration re-exports these packs + let c = LoadClass::era(LoadClass::V2, &eb, &allowed); + let e = c.era.unwrap(); + let words: Vec = j["era"]["seed_words"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(e.words.to_vec(), words, "{pack}: era seed words"); + assert_eq!(e.width_words as u64, j["era"]["width_words"].as_u64().unwrap(), "{pack}: width"); + assert_eq!(e.stride_mul, hex32(&j["era"]["stride_mul"]), "{pack}: stride mul"); + assert_eq!(e.stride_rot as u64, j["era"]["stride_rot"].as_u64().unwrap(), "{pack}: stride rot"); + let pos: Vec = j["era"]["interleave"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u8).collect(); + assert_eq!(e.pos.to_vec(), pos, "{pack}: interleave"); + c +} + +fn era_epoch(pack: &str) -> &'static Epoch { + static E: OnceLock> = OnceLock::new(); + let all = E.get_or_init(|| { + ERA_PACKS + .iter() + .map(|p| { + let j = era_json(p, "program.json"); + let seed = j["seed"].as_str().unwrap(); + let seed_bytes = unhex(&j["seed_bytes"]); + let day_bytes = unhex(&j["dataset"]["day_bytes"]); + assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard"); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + let class = era_class(p); + assert_eq!(log2, igneum_pow::verify::DEFAULT_DATASET_LOG2); + let eb = unhex(&j["era_seed_bytes"]); + let program = generate_era(seed, &seed_bytes, LoadClass::V2, &eb, &V3_ALLOWED); + assert_eq!(program.class, class, "{p}: the pack's era class"); + assert_eq!(program.generator, GENERATOR_VERSION_V3); + assert_eq!(program.era_bytes.as_deref(), Some(&eb[..])); + let mut dataset = DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2); + dataset.key_bytes = day_bytes; + (p.to_string(), Epoch { program, dataset }) + }) + .collect() + }); + &all.iter().find(|(n, _)| n == pack).unwrap().1 +} + +/// program.json of an era pack: generator, attempt, id, class, every instruction with width, win and off, and the +/// program passes the acceptance rule. +#[test] +fn era_program_json_matches() { + for pack in ERA_PACKS { + let j = era_json(pack, "program.json"); + let p = &era_epoch(pack).program; + assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION_V3, "{pack}: a class v3 pack"); + assert_eq!(j["program_class"].as_str().unwrap(), "v3"); + assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt, "{pack}: attempt"); + assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id"); + assert_eq!(j["load_class"].as_str().unwrap(), p.class.name(), "{pack}: class"); + assert_eq!(j["bytes_per_hash"].as_u64().unwrap() as usize, p.bytes_per_hash()); + assert_eq!(p.loads_per_hash(), 8 * LOAD_SLOTS); + assert!(accept::check(p).is_ok(), "{pack}: acceptance"); + let instrs = j["instructions"].as_array().unwrap(); + assert_eq!(instrs.len(), p.instrs.len()); + for (k, (ins, ji)) in p.instrs.iter().zip(instrs).enumerate() { + assert_eq!(Op::from_name(ji["op"].as_str().unwrap()).unwrap(), ins.op, "{pack} #{k} op"); + assert_eq!(ji["dst"].as_u64().unwrap(), ins.dst as u64); + assert_eq!(ji["src"].as_u64().unwrap(), ins.src as u64); + assert_eq!(hex32(&ji["imm"]), ins.imm); + assert_eq!(ji["width"].as_u64().unwrap(), ins.width as u64, "{pack} #{k} width"); + assert_eq!(ji["win"].as_u64().unwrap(), ins.win as u64, "{pack} #{k} win"); + assert_eq!(ji["off"].as_u64().unwrap(), ins.off as u64, "{pack} #{k} off"); + if ins.op == Op::Load { + assert_eq!(ins.width, p.class.era.unwrap().width_words); + assert!(ins.win <= 2 && (ins.off as u32) < (1u32 << ins.win)); + } + } + } +} + +/// The six era packs are the same program seed under six draws: the instruction lists agree, the widths and layouts +/// follow the draw, and the dataset words differ from the linear layout exactly when the interleave is not linear. +#[test] +fn era_packs_share_the_program_and_differ_in_layout() { + let linear = &epoch("igneum-devnet-v4-epoch0").dataset; + for pack in ERA_PACKS { + let e = era_epoch(pack); + assert_eq!(e.program.seed_bytes, epoch("igneum-devnet-v4-epoch0").program.seed_bytes, "{pack}: the devnet seed"); + let strip = |p: &igneum_pow::generator::Program| { + p.instrs.iter().map(|i| (i.op, i.dst, i.src, i.src2, i.imm, i.imm2, i.rot, i.bit, i.mask, i.win, i.off)).collect::>() + }; + assert_eq!(strip(&e.program), strip(&era_epoch("era-0").program), "{pack}: same stream as era-0"); + let l = e.program.class.layout(); + let same_at_1 = (0..64u32).all(|w| e.dataset_word(w * 977 + 1) == linear.word(w * 977 + 1)); + assert_eq!(same_at_1, l.is_linear(), "{pack}: layout {:?}", l.pos); + assert_eq!(e.dataset_word(0), linear.word(0), "{pack}: word 0 is item 0 word 0 in every layout"); + // the chain's shared day cache serves every era: the day's dataset source is the pinned pack's, bit for bit + assert_eq!(e.dataset.key, linear.key); + assert_eq!(e.dataset.memhard().unwrap().cache.fnv1a64(), linear.memhard().unwrap().cache.fnv1a64()); + } +} + +#[test] +fn era_dataset_words_and_vectors_match() { + for pack in ERA_PACKS { + let v = era_json(pack, "vectors.json"); + let e = era_epoch(pack); + let ds = &e.dataset; + let head: Vec = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect(); + for (i, h) in head.iter().enumerate() { + assert_eq!(e.dataset_word(i as u32), *h, "{pack}: dataset[{i}]"); + } + assert_eq!(e.dataset_word(ds.mask), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]"); + for s in v["dataset_samples"].as_array().unwrap() { + let idx = s["index"].as_u64().unwrap() as u32; + assert_eq!(e.dataset_word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]"); + } + assert_eq!(ds.memhard().unwrap().cache.fnv1a64(), hex64(&v["cache_fnv1a64"])); + let mut n = 0; + for w in v["warps"].as_array().unwrap() { + let base = w["base_nonce"].as_u64().unwrap() as u32; + let expected: Vec = w["expected"].as_array().unwrap().iter().map(hex64).collect(); + let got = e.hash_warp(base); + for lane in 0..32 { + assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}"); + n += 1; + } + assert_eq!(e.hash(base + 7), expected[7]); + } + assert_eq!(n, 96, "{pack}"); + } +} + +/// Every emitted file of every era pack matches the emitters byte for byte, the export reproduces vectors.json and +/// vectors.h, and every dataset load in every hash kernel has the one era form (no plain `ds[rN & mask]` remains). +#[test] +fn era_emitted_sources_match_and_loads_have_the_era_form() { + for pack in ERA_PACKS { + let e = era_epoch(pack); + let day = era_json(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string(); + let source = era_json(pack, "vectors.json")["source"].as_str().unwrap().to_string(); + let out = export_pack(e, &day, &source); + for (name, text) in &out.files { + let want = era_read(pack, name); + assert!(text == &want, "{pack}/{name} differs from the emitter"); + } + let mut on_disk: Vec = std::fs::read_dir(era_packs_dir().join(pack)) + .unwrap() + .map(|d| d.unwrap().file_name().to_string_lossy().to_string()) + .filter(|n| !n.starts_with('.') && n != "seeds.txt") + .collect(); + on_disk.sort(); + let mut want: Vec = out.files.iter().map(|(n, _)| n.clone()).collect(); + want.sort(); + assert_eq!(on_disk, want, "{pack}: the pack holds the export's files and seeds.txt only"); + let era = e.program.class.era.unwrap(); + let mul = format!("0x{:08x}u", era.stride_mul); + for (file, mask) in [ + ("kernel.cu", "mask"), + ("kernel_bound.cu", "mask"), + ("kernel.cl", "mask"), + ("kernel_bound.cl", "mask"), + ("program.metal", "MASK"), + ("program_bound.metal", "MASK"), + ] { + let text = era_read(pack, file); + let kernels = if file.starts_with("kernel_bound") || file == "kernel.cu" || file == "kernel.cl" || file.starts_with("program") { 1 } else { 1 }; + // kernel_bound.cl carries igneum_hash and igneum_hash_bound: two kernels + let kernels = if file == "kernel_bound.cl" { 2 } else { kernels }; + let era_form: usize = text + .lines() + .filter(|l| l.contains("rotl_imm(r") && l.contains(&format!(" * {mul}, {}u) & ", era.stride_rot)) && l.contains(&format!(") & {mask}"))) + .filter(|l| l.contains("ds[") || l.contains("dataset[") || l.contains("b_ = ")) + .count(); + assert_eq!(era_form, LOAD_SLOTS * kernels, "{pack}/{file}: {} loads of the era form", LOAD_SLOTS * kernels); + let plain = text.lines().filter(|l| l.contains("ds[r") || l.contains("dataset[r")).count(); + assert_eq!(plain, 0, "{pack}/{file}: a load without the era form"); + } + // the layout helpers appear exactly when the layout is not linear + let mh = era_read(pack, "memhard.h"); + assert_eq!(mh.contains("mh_addr("), !era.layout().is_linear(), "{pack}: memhard.h layout helpers"); + } +} + +/// An era pack's dataset is a prefix at every size of at least 2^16 words: the 2^20-word source gives the pack's +/// words below 2^20. +#[test] +fn era_dataset_is_a_prefix_at_smaller_sizes() { + for pack in ["era-1", "era-3"] { + let e = era_epoch(pack); + let small = DatasetSource::from_key(e.dataset.key, DatasetMode::MemoryHard, 20); + let l = e.program.class.layout(); + for w in [0u32, 1, 2, 3, 16, 255, 4096, 65_535, 65_536, 0x000f_ffff] { + assert_eq!(small.word_at(l, w), e.dataset_word(w), "{pack}: w {w}"); + } + } +} + +// --------------------------------------------------------------------------------------------------------- +// Hot-table experiment (5 October 2026, docs/plans/hot-table.md): the five packs under proto-cuda/packs-ca2-hot/ are +// pinned the same way (program, vectors, every emitted file byte for byte), plus the hot table's fingerprint and the +// one-form load check: exactly 16 - k masked dataset loads and exactly k hot loads in every hash kernel. +// --------------------------------------------------------------------------------------------------------- + +const HOT_PACKS: [&str; 8] = ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "hot32k4a", "hot64k4a", "hot96k4a"]; + +fn hot_packs_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-hot") +} + +fn hread(pack: &str, file: &str) -> String { + let p = hot_packs_dir().join(pack).join(file); + std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display())) +} + +fn hjson(pack: &str, file: &str) -> Value { + serde_json::from_str(&hread(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")) +} + +/// The epoch of a hot pack from program.json alone: the class from `load_class`, the program from the seed bytes, +/// the dataset from the day bytes, the hot table from the seed bytes (what `Epoch::from_seed_bytes_class` does). +fn hepoch(pack: &str) -> &'static Epoch { + static E: OnceLock> = OnceLock::new(); + let all = E.get_or_init(|| { + HOT_PACKS + .iter() + .map(|p| { + let j = hjson(p, "program.json"); + let seed = j["seed"].as_str().unwrap(); + let seed_bytes = unhex(&j["seed_bytes"]); + let day_bytes = unhex(&j["dataset"]["day_bytes"]); + assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard"); + let class = LoadClass::parse(j["load_class"].as_str().unwrap()).unwrap(); + assert_eq!(class.name(), *p, "the pack directory is the class name"); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + let program = generate_from_seed_bytes_class(seed, &seed_bytes, class); + let mut dataset = + DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2); + dataset.key_bytes = day_bytes; + dataset.attach_hot_for(&program); + (p.to_string(), Epoch { program, dataset }) + }) + .collect() + }); + &all.iter().find(|(n, _)| n == pack).unwrap().1 +} + +fn hassert_same_text(pack: &str, file: &str, got: &str) { + let want = hread(pack, file); + if got != want { + let (gl, wl): (Vec<&str>, Vec<&str>) = (got.lines().collect(), want.lines().collect()); + for i in 0..gl.len().max(wl.len()) { + let g = gl.get(i).copied().unwrap_or(""); + let w = wl.get(i).copied().unwrap_or(""); + if g != w { + panic!("{pack}/{file} differs at line {}:\n pack: {w}\n rust: {g}", i + 1); + } + } + panic!("{pack}/{file} differs only in trailing bytes (len {} vs {})", got.len(), want.len()); + } +} + +#[test] +fn hot_packs_program_and_vectors() { + for pack in HOT_PACKS { + let e = hepoch(pack); + let p = &e.program; + let j = hjson(pack, "program.json"); + let h = p.class.hot.unwrap(); + assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION); + assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt); + assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id"); + let dataset_slots = if h.added { 16 } else { 16 - h.k as usize }; + assert_eq!(j["loads_per_hash"].as_u64().unwrap() as usize, (dataset_slots + h.k as usize) * 8); + assert_eq!(j["hot_table"]["mb"].as_u64().unwrap(), h.mb as u64); + assert_eq!(j["hot_table"]["slots"].as_u64().unwrap(), h.k as u64); + assert_eq!(j["hot_table"]["dataset_slots"].as_u64().unwrap() as usize, dataset_slots); + assert_eq!(j["hot_table"]["words"].as_u64().unwrap() as u32, p.hot_words()); + assert_eq!(j["op_mix"]["hot"].as_u64().unwrap(), h.k as u64, "{pack}: k hot instructions"); + assert_eq!(j["op_mix"]["load"].as_u64().unwrap() as usize, dataset_slots); + assert_eq!(p.items_per_warp(), dataset_slots * 8 * 32); + assert!(accept::check(p).is_ok(), "{pack}: passes the acceptance rule"); + let v2 = &epoch("igneum-genesis-mh").program; + if !h.added { + // replaced form: the version 2 genesis program with k loads redirected (attempt 0 on both) + assert_eq!(p.attempt, v2.attempt); + for (a, b) in p.instrs.iter().zip(v2.instrs.iter()) { + if a.op == Op::Hot { + assert_eq!(b.op, Op::Load); + } else { + assert_eq!(a, b); + } + } + } else { + // added form: 16 + k load slots, so another slot draw and another program; 16 dataset loads stay + assert_ne!(p.instrs, v2.instrs); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16); + assert!(p.instrs.iter().all(|i| i.width == 1)); + } + // the hot table: the pack's head, last line and fingerprint + let v = hjson(pack, "vectors.json"); + let t = e.dataset.hot.as_ref().unwrap(); + assert_eq!(t.n_words(), p.hot_words()); + let head: Vec = v["hot_head"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(&t.words()[..16], &head[..]); + let last: Vec = v["hot_last_line"].as_array().unwrap().iter().map(hex32).collect(); + assert_eq!(&t.words()[t.words().len() - 16..], &last[..]); + assert_eq!(t.fnv1a64(), hex64(&v["hot_fnv1a64"]), "{pack}: hot_fnv1a64"); + assert_eq!(t.key, igneum_pow::memhard::hot_key(&p.seed_bytes)); + // the cache is the day's, unchanged by the class + assert_eq!(e.dataset.memhard().unwrap().cache.fnv1a64(), 0x48c4f5bf24166b2e); + // 96 vectors + let warps = v["warps"].as_array().unwrap(); + assert_eq!(warps.len(), 3); + for w in warps { + let base = w["base_nonce"].as_u64().unwrap() as u32; + let expected: Vec = w["expected"].as_array().unwrap().iter().map(hex64).collect(); + let got = e.hash_warp(base); + for lane in 0..32 { + assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}"); + } + assert_eq!(e.hash(base + 5), expected[5]); + } + // the dataset words are the day's + let head: Vec = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect(); + for (i, hd) in head.iter().enumerate() { + assert_eq!(e.dataset.word(i as u32), *hd); + } + } + // the same k at three sizes: identical programs, three fingerprints, three vector sets + let a = hepoch("hot32k4"); + let b = hepoch("hot64k4"); + let c = hepoch("hot96k4"); + assert_eq!(a.program.instrs, b.program.instrs); + assert_eq!(b.program.instrs, c.program.instrs); + assert_ne!(a.hash_warp(0), b.hash_warp(0)); + assert_ne!(b.hash_warp(0), c.hash_warp(0)); +} + +#[test] +fn hot_packs_emitted_sources_and_load_forms() { + for pack in HOT_PACKS { + let e = hepoch(pack); + let p = &e.program; + let k = p.class.hot.unwrap().k as usize; + let dataset_loads = p.class.dataset_slots(); + let day = hjson(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string(); + let mp = &e.dataset.memhard().unwrap().params; + hassert_same_text(pack, "kernel.cu", &cuda_kernel(p, Some(mp))); + hassert_same_text(pack, "kernel_bound.cu", &cuda_kernel_bound(p, Some(mp))); + hassert_same_text(pack, "program.metal", &metal_program(p, e.dataset.log2_words, LoadSource::Stored)); + hassert_same_text(pack, "program_bound.metal", &metal_program_bound(p, e.dataset.log2_words)); + hassert_same_text(pack, "kernel.cl", &opencl_kernel(p, Some(mp))); + hassert_same_text(pack, "kernel_bound.cl", &opencl_kernel_bound(p, Some(mp))); + hassert_same_text(pack, "program.h", &program_header(p, &day, &e.dataset)); + hassert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp)); + hassert_same_text(pack, "memhard.metal", &metal_memhard_for(p, mp)); + assert_ne!(metal_memhard_for(p, mp), metal_memhard(mp), "{pack}: the hot fill kernel is in memhard.metal"); + let got = program_json(p, &day, &e.dataset); + hassert_same_text(pack, "program.json", &got); + let _: Value = serde_json::from_str(&got).expect("program.json is valid JSON"); + let v = hjson(pack, "vectors.json"); + let out = export_pack(e, &day, v["source"].as_str().unwrap()); + let file = |name: &str| -> &str { &out.files.iter().find(|(n, _)| n == name).unwrap().1 }; + hassert_same_text(pack, "vectors.json", file("vectors.json")); + hassert_same_text(pack, "vectors.h", file("vectors.h")); + assert_eq!(out.files.len(), 12); + // One form per dialect, exactly 16 - k masked dataset loads and k hot loads in every hash kernel; the fill + // kernel is present once per source that builds the table. + for (file, load, masked, hot) in [ + ("kernel.cu", "ds[r", " & mask]", "hot[__umulhi(r"), + ("kernel_bound.cu", "ds[r", " & mask]", "hot[__umulhi(r"), + ("program.metal", "dataset[r", " & MASK]", "hot[mulhi(r"), + ("program_bound.metal", "dataset[r", " & MASK]", "hot[mulhi(r"), + ("kernel.cl", "ds[r", " & mask]", "hot[mul_hi(r"), + ] { + let text = hread(pack, file); + assert_eq!(text.matches(load).count(), dataset_loads, "{pack}/{file}: {dataset_loads} dataset loads"); + assert_eq!(text.matches(masked).count(), dataset_loads, "{pack}/{file}: masked loads"); + assert_eq!(text.matches(hot).count(), k, "{pack}/{file}: {k} hot loads"); + assert!(text.contains(&format!("#define HOT_WORDS 0x{:08x}u", p.hot_words())), "{pack}/{file}: HOT_WORDS literal"); + } + // kernel_bound.cl carries both kernels + let text = hread(pack, "kernel_bound.cl"); + assert_eq!(text.matches("hot[mul_hi(r").count(), 2 * k); + assert_eq!(text.matches("ds[r").count(), 2 * dataset_loads); + for file in ["kernel.cu", "kernel.cl", "kernel_bound.cl", "memhard.metal"] { + assert_eq!(hread(pack, file).matches("igneum_hot_fill(").count(), 1, "{pack}/{file}: one hot fill kernel"); + } + assert_eq!(hread(pack, "memhard.h").matches("void ht_segment(").count(), 1); + let ph = hread(pack, "program.h"); + assert!(ph.contains(&format!("#define IGNEUM_HOT_MB {}", p.class.hot.unwrap().mb))); + assert!(ph.contains(&format!("#define IGNEUM_HOT_SLOTS {k}"))); + assert!(ph.contains("igneum_launch_hot_fill(")); + } +} + +// --------------------------------------------------------------------------------------------------------------- +// Class v5 (docs/design/class-v5-stored-state.md, 7 October 2026): the pinned pack under proto-cuda/packs-ca3-v5/ +// --------------------------------------------------------------------------------------------------------------- + +fn v5_packs_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca3-v5") +} + +fn v5_read(pack: &str, file: &str) -> String { + std::fs::read_to_string(v5_packs_dir().join(pack).join(file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")) +} + +fn v5_json(pack: &str, file: &str) -> Value { + serde_json::from_str(&v5_read(pack, file)).unwrap() +} + +/// The epoch of a pinned class v4 or v5 pack of the string seed and day: the program through the seam, the dataset at +/// the day-0 size, and for a v5 pack the leaves of its `state.igsd1` (the devnet's state stream of 7 October 2026, +/// node 1's exec snapshot at chain block 159,357: 93 records, root 0x1c583d35...). +fn v5_epoch(pack: &str) -> Epoch { + let j = v5_json(pack, "program.json"); + let seed = j["seed"].as_str().unwrap(); + let class = ProgramClass::parse(j["program_class"].as_str().unwrap()).unwrap(); + let program = generate_from_seed_bytes_program_class(seed, seed.as_bytes(), class, None); + let day = j["dataset"]["day"].as_str().unwrap(); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + let shape = Shape::for_class(&program.class); + let mut dataset = DatasetSource::new_shape(day, DatasetMode::MemoryHard, log2, shape); + if class == ProgramClass::V5 { + let stream = igneum_pow::StateStream::read_file(&v5_packs_dir().join(pack).join("state.igsd1")).unwrap(); + dataset = dataset.with_leaves(std::sync::Arc::new(igneum_pow::StateLeaves::from_stream(&stream, log2))); + } + Epoch { program, dataset } +} + +/// The class v5 pack is the class v4 program of the same seed over the state leaves: generator 5 and +/// `program_id(5, seed, attempt)`, the base program, the shadow block and the dataset's cache equal to the v4 pack's, +/// every vector and dataset word different, every emitted file byte for byte what the crate exports (leaves.bin +/// included, its FNV in program.h), and the hash kernel text of kernel.cu equal to the v4 pack's but for the build +/// kernel and the class lines. The known-failed case first: the v4 control pack's vectors under the v5 epoch agree on +/// no lane. +#[test] +fn v5_pack_is_the_v4_program_over_the_state_leaves() { + let e5 = v5_epoch("v5-genesis"); + let e4 = v5_epoch("v4-genesis"); + let v4 = v5_json("v4-genesis", "vectors.json"); + // the known-failed case: the v4 pack's hashes are not the v5 epoch's on any lane + let out4: Vec = v4["warps"][0]["expected"].as_array().unwrap().iter().map(hex64).collect(); + let got5 = e5.hash_warp(0); + assert_eq!(out4.iter().zip(got5.iter()).filter(|(a, b)| a == b).count(), 0, "0 of 32 lanes of the v4 pack agree with the v5 epoch"); + assert_eq!(e4.hash_warp(0).to_vec(), out4, "the v4 control pack is the v4 epoch"); + // the program: v4's draw, generator 5, the state in the class and the id + assert_eq!(e5.program.generator, igneum_pow::GENERATOR_VERSION_V5); + assert_eq!(e5.program.class, igneum_pow::V5_CLASS); + assert_eq!(e5.program.class.name(), "mx8+sh256x27+state"); + assert_eq!(e5.program.instrs, e4.program.instrs); + assert_eq!(e5.program.shadow, e4.program.shadow); + assert_eq!((e5.program.seed, e5.program.attempt), (e4.program.seed, e4.program.attempt)); + assert_eq!(e5.program.program_id(), igneum_pow::generator::program_id(igneum_pow::GENERATOR_VERSION_V5, &e5.program.seed, e5.program.attempt)); + assert_ne!(e5.program.program_id(), e4.program.program_id()); + // the dataset: the same cache, other items + let m5 = e5.dataset.memhard().unwrap(); + let m4 = e4.dataset.memhard().unwrap(); + assert_eq!(m5.cache.fnv1a64(), m4.cache.fnv1a64(), "one day cache"); + assert!(m5.shape().state && !m4.shape().state); + let leaves = e5.dataset.leaves().unwrap(); + assert_eq!((leaves.n(), leaves.records_total, leaves.sampled), (93, 93, false)); + assert_ne!(e5.dataset_word(0), e4.dataset_word(0)); + // every file as the crate exports it, leaves.bin included + for pack in ["v4-genesis", "v5-genesis"] { + let e = if pack == "v5-genesis" { &e5 } else { &e4 }; + let v = v5_json(pack, "vectors.json"); + let out = export_pack(e, v["day"].as_str().unwrap(), v["source"].as_str().unwrap()); + for (name, text) in &out.files { + assert_eq!(v5_read(pack, name), *text, "{pack}/{name} differs from the export"); + } + for (name, bytes) in &out.binaries { + assert_eq!(std::fs::read(v5_packs_dir().join(pack).join(name)).unwrap(), *bytes, "{pack}/{name}"); + } + assert_eq!(out.binaries.len(), (pack == "v5-genesis") as usize); + } + let h5 = v5_read("v5-genesis", "program.h"); + assert!(h5.contains("#define IGNEUM_PROGRAM_CLASS \"v5\"") && h5.contains("#define IGNEUM_STATE_LEAVES 93") && h5.contains(&format!("#define IGNEUM_STATE_LEAVES_FNV64 {}", igneum_pow::emit::hex64(leaves.fnv1a64())))); + assert!(h5.contains("igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems)")); + // the hash kernel text is class v4's byte for byte; the build kernel and the class lines are what differ + let hash_text = |s: &str| { + let a = s.find("__global__ void igneum_hash(").unwrap(); + let b = s.find("// Host-side launch wrappers").unwrap(); + s[a..b].to_string() + }; + let k5 = v5_read("v5-genesis", "kernel.cu"); + let k4 = v5_read("v4-genesis", "kernel.cu"); + assert_eq!(hash_text(&k5), hash_text(&k4), "the hash kernel is class v4's"); + assert!(k5.contains("mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s)") && !k4.contains("mh_leaf")); + let mh5 = v5_read("v5-genesis", "memhard.h"); + assert!(mh5.contains("s[i] ^= leaf[i]"), "the leaf XOR before the first mixer"); +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/tests/recheck.rs b/tools/attack/adv-accept-v5/igneum-pow/tests/recheck.rs new file mode 100644 index 000000000..9b2f4f768 --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/tests/recheck.rs @@ -0,0 +1,143 @@ +//! The miner's CPU re-check against the worker's reference path, class v3 and class v4 (6 October 2026). +//! +//! The fleet's 10-member pool run (18:14Z to 18:24Z): 0 shares accepted in ten minutes, every GPU answer refused as +//! `WORKER MISMATCH`, because the pool fork's miner re-hashed each share with a fixed class v2 program while the 0.3.14 +//! CUDA worker hashed the class v3 program of the epoch. The rule since: the re-check builds its program through the +//! chain seam (`Epoch::chain_program` and `Epoch::chain_dataset_day`, what the node's `IgneumEngine` calls) from the +//! template's `(epoch seed, day, program class, era seed)`, never from a fixed class. +//! +//! This test pins that seam against the worker's reference for both classes the packs carry: the program the pack's +//! kernels were emitted from (program.json, generator 3 or 4, with its id) and the pack's 96 vectors; then one known +//! nonce through the worker's path (`block_init_words` plus `interpret_warp_init` on the pack's program) and through +//! the re-check's path (the same on the seam's program and dataset) must agree, and the other class's program and +//! the fixed-v2 program of the same seed must disagree (the mismatch the fleet saw). + +use igneum_pow::bind::{block_init_words, lane_nonce, unhex}; +use igneum_pow::verify::DatasetSource; +use igneum_pow::{interpret_warp_init, DatasetMode, Epoch, ProgramClass, Shape}; +use serde_json::Value; +use std::path::PathBuf; + +fn pack_dir(rel: &str) -> PathBuf { + let p = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda").join(rel); + assert!(p.join("program.json").is_file(), "pack {} is missing; this test never skips", p.display()); + p +} + +fn json(dir: &std::path::Path, f: &str) -> Value { + serde_json::from_str(&std::fs::read_to_string(dir.join(f)).unwrap_or_else(|e| panic!("{}: {e}", dir.join(f).display()))).unwrap() +} + +fn hex64(v: &Value) -> u64 { + u64::from_str_radix(v.as_str().unwrap().trim_start_matches("0x"), 16).unwrap() +} + +struct PackRef { + class: ProgramClass, + seed_label: String, + seed_bytes: Vec, + era: Vec, + day_bytes: Vec, + log2: u32, + /// The worker's reference: the pack's program and day dataset, checked against the pack's vectors + reference: Epoch, +} + +/// The pack as the worker sees it: program.json's program (generator, id and class as recorded) over the pack's +/// day dataset, trusted only after its 96 vectors reproduce. +fn load(rel: &str, want: ProgramClass) -> PackRef { + let dir = pack_dir(rel); + let j = json(&dir, "program.json"); + let class = match j["program_class"].as_str().unwrap() { + "v3" => ProgramClass::V3, + "v4" => ProgramClass::V4, + other => panic!("{rel}: class {other}"), + }; + assert_eq!(class, want, "{rel}: the pack's class"); + let seed_label = j["seed"].as_str().unwrap().to_string(); + let seed_bytes = unhex(j["seed_bytes"].as_str().unwrap()).unwrap(); + let era = unhex(j["era_seed_bytes"].as_str().unwrap()).unwrap(); + let day_bytes = unhex(j["dataset"]["day_bytes"].as_str().unwrap()).unwrap(); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + let program = igneum_pow::generate_from_seed_bytes_program_class(&seed_label, &seed_bytes, class, Some(&era)); + assert_eq!(program.generator, j["generator"].as_u64().unwrap() as u32, "{rel}: generator"); + assert_eq!(program.generator, class.generator_version(), "{rel}: the generator is the class's"); + assert_eq!(program.program_id(), hex64(&j["program_id"]), "{rel}: program id"); + let shape = Shape::for_class(&program.class); + let mut dataset = DatasetSource::from_key_shape(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2, shape); + dataset.key_bytes = day_bytes.clone(); + let reference = Epoch { program, dataset }; + let v = json(&dir, "vectors.json"); + let warps = v["warps"].as_array().unwrap(); + assert_eq!(warps.len(), 3, "{rel}: three vector warps"); + for w in warps { + let base = w["base_nonce"].as_u64().unwrap() as u32; + let got = reference.hash_warp(base); + for (lane, e) in w["expected"].as_array().unwrap().iter().enumerate() { + assert_eq!(got[lane], hex64(e), "{rel}: vector base {base} lane {lane}"); + } + } + PackRef { class, seed_label, seed_bytes, era, day_bytes, log2, reference } +} + +/// The re-check's epoch: the chain seam the node engine and the miner call, from the template's four values. +/// The packs are day-0 caches, so days since genesis is 0 and the genesis dataset size is the pack's. +fn recheck_epoch(p: &PackRef) -> Epoch { + Epoch { program: Epoch::chain_program(&p.seed_bytes, Some(&p.era), p.class, &p.seed_label), dataset: Epoch::chain_dataset_day(&p.day_bytes, p.class, 0, p.log2) } +} + +const PREHASH: [u8; 32] = [0x5a; 32]; +const NONCE: u64 = 0x1234_5678_0000_0bb7; // high word 0x12345678, lane 0x0bb7 (warp base 0x0ba0, lane 23) + +/// The bound lane hash of one nonce on an epoch, the way a worker computes it (init words from the prehash and the +/// nonce's high word, the warp at the lane's base) and the way `EpochRef::hash_bound` re-checks it. +fn bound(e: &Epoch, prehash: &[u8; 32], nonce: u64) -> u64 { + let init = block_init_words(prehash, nonce); + let lane = lane_nonce(nonce); + interpret_warp_init(&e.program, &init, lane & !31, &e.dataset).hashes[(lane & 31) as usize] +} + +#[test] +fn cpu_recheck_equals_the_worker_reference_for_class_v3_and_class_v4() { + let v3 = load("packs-ca2-mixer/mx8-devnet-epoch0", ProgramClass::V3); + let v4 = load("packs-ca3-v4/v4-devnet-epoch0", ProgramClass::V4); + assert_eq!(v3.seed_bytes, v4.seed_bytes, "the two packs share the epoch seed (devnet genesis), so only the class differs"); + assert_eq!(v3.era, v4.era); + let mut hashes = Vec::new(); + for p in [&v3, &v4] { + let re = recheck_epoch(p); + assert_eq!(re.program.program_id(), p.reference.program.program_id(), "{:?}: the seam's program is the pack's", p.class); + assert_eq!(re.program.generator, p.reference.program.generator); + assert_eq!(re.program.instrs.len(), p.reference.program.instrs.len()); + let worker = bound(&p.reference, &PREHASH, NONCE); + let cpu = bound(&re, &PREHASH, NONCE); + assert_eq!(cpu, worker, "{:?}: CPU re-check {cpu:016x} against the worker's reference {worker:016x}", p.class); + // the same nonce through a second prehash, so the agreement is not one lucky lane + let (w2, c2) = (bound(&p.reference, &[0xa5; 32], NONCE ^ 0x1f), bound(&re, &[0xa5; 32], NONCE ^ 0x1f)); + assert_eq!(c2, w2, "{:?}: second nonce", p.class); + hashes.push(cpu); + } + assert_ne!(hashes[0], hashes[1], "class v3 and class v4 programs of one seed hash a nonce differently"); + // the fleet's mismatch: a fixed class v2 program on the class v3 epoch's seed + let v2 = Epoch { program: Epoch::chain_program(&v3.seed_bytes, None, ProgramClass::V2, &v3.seed_label), dataset: Epoch::chain_dataset_day(&v3.day_bytes, ProgramClass::V2, 0, v3.log2) }; + assert_ne!(bound(&v2, &PREHASH, NONCE), hashes[0], "a class v2 re-check refuses a class v3 share"); + assert_ne!(bound(&v2, &PREHASH, NONCE), hashes[1], "a class v2 re-check refuses a class v4 share"); +} + +/// The generator-4 program id rule (counter-asic-3-status.md, the blocker closed 6 October 2026): the v4 program of +/// a seed carries another id than the v3 program of the same seed, so a stale worker across the activation sees the +/// mismatch, while v2 and v3 ids stay as the packs pinned them. +#[test] +fn program_ids_differ_between_class_v3_and_class_v4_of_one_seed() { + let v3 = load("packs-ca2-mixer/mx8-devnet-epoch0", ProgramClass::V3); + let v4 = load("packs-ca3-v4/v4-devnet-epoch0", ProgramClass::V4); + assert_eq!(v3.reference.program.program_id(), 0x73bc_bfe8_ccf9_88f1, "the v3 control's id as pinned"); + // the amended class v4 (AP-F8-1, AP-F8-3): sub-version 3 (object byte 7; the acceptance executes the shadow block) + // is the pinned id below; c120d7963abdcd96 (the 6 October stream, byte 4), 1a4230699a6b9c60 (sub-version 1, the + // one-writer rule, 0.3.20's and 0.3.21's byte 5) and a788661687db4bb3 (sub-version 2, never shipped) must differ + assert_ne!(v4.reference.program.program_id(), 0xc120_d796_3abd_cd96, "the pre-amendment v4 id must differ"); + assert_ne!(v4.reference.program.program_id(), 0x1a42_3069_9a6b_9c60, "the sub-version-1 id must differ"); + assert_ne!(v4.reference.program.program_id(), 0xa788_6616_87db_4bb3, "the sub-version-2 id must differ"); + assert_eq!(v4.reference.program.program_id(), 0xa785_0016_87d8_688a, "the sub-version-3 v4 id as pinned"); + assert_ne!(v3.reference.program.program_id(), v4.reference.program.program_id()); +} diff --git a/tools/attack/adv-accept-v5/igneum-pow/tests/scratch.rs b/tools/attack/adv-accept-v5/igneum-pow/tests/scratch.rs new file mode 100644 index 000000000..5ad1c873d --- /dev/null +++ b/tools/attack/adv-accept-v5/igneum-pow/tests/scratch.rs @@ -0,0 +1,769 @@ +//! Soundness tests of layer 3 of `docs/plans/counter-asic-2.md`: the per-warp scratch with read-modify-writes +//! (variant 5 of the read-width experiment, `LoadClass::scratch(k, kb)`). Analysis and results: +//! `docs/analysis/scratch-soundness.md`. Every test is parametric over the class's slot count +//! (`scratch_slots_per_lane()`), so the 32 and 128 KiB geometries and any later one run the same checks. +//! +//! What runs under plain `cargo test`: +//! 1. `rewrite_is_a_bijection_of_the_fold_value`, `fill_is_a_bijection_of_the_nonce`: the written words as +//! functions (question 1). +//! 2. `written_words_unbiased_and_rehit_rates`: bit bias of every written word over 2^11 units x 3 seeds per class +//! (the TESTS.md section 3 shape), and the measured slot re-hit rate against the birthday formula (question 2). +//! 3. `edge_programs_match_the_hand_model`: hand-built programs that drive every read-modify-write of a hash to +//! slot 0, slot MASK, through out-of-range registers, to one slot per lane, alternating two slots, and 16 +//! read-modify-writes per iteration on one slot; the interpreter against an independent hand model, and the +//! hand model shown to have teeth (question 3, CPU half). +//! 4. `scr_packs_regenerate_and_pass_the_static_scratch_check`: every emitted kernel of every scr pack under +//! `proto-cuda/packs-readwidth` regenerates from its program.json and passes the static scratch-mask check; +//! the check is shown to fail on four deliberate breaks (question 4). +//! 5. `fuzz_scr_programs_cpu`: 200 generated scratch programs over the six classes, generator contract on every +//! instruction, 4 units each at base nonces across the 32-bit range including the wrap; with +//! `IGNEUM_SCRATCH_PACKS_OUT=` it also writes the packs (and the edge packs) for the Metal runs of +//! `proto-metal/packbench` (question 3 GPU half, question 4, `TESTS.md` section 9 shape). + +use igneum_pow::emit::{ + cuda_kernel, cuda_kernel_bound, export_pack, metal_program, metal_program_bound, opencl_kernel, + opencl_kernel_bound, vectors_json, LoadSource, +}; +use igneum_pow::generator::{ + generate_class, generate_from_seed_bytes_class, Instr, LoadClass, Op, Program, GENERATOR_VERSION, INSTR_COUNT, + ITERATIONS, LANES, +}; +use igneum_pow::seed::{seed_words_from_bytes, SplitMix64}; +use igneum_pow::verify::{ + fold_words, interpret_warp_scratch, scratch_fill, scratch_rewrite, splitmix32, DatasetMode, DatasetSource, + Epoch, ScratchEvent, FOLD_MUL, FOLD_ROT, +}; +use serde_json::Value; +use std::collections::HashMap; +use std::path::PathBuf; + +/// The classes under study: the two capped geometries (32 and 128 KiB per warp: 64 and 256 slots per lane) at the +/// RMW shares the readwidth branch measures. +const CLASSES: [&str; 6] = ["scr2k32", "scr4k32", "scr8k32", "scr2k128", "scr4k128", "scr8k128"]; + +fn class(name: &str) -> LoadClass { + LoadClass::parse(name).unwrap_or_else(|| panic!("class {name}")) +} + +// --------------------------------------------------------------------------------------------------------------- +// 1. The written words as functions (question 1) +// --------------------------------------------------------------------------------------------------------------- + +/// For a fixed slot content `w`, each of the three rewritten words is a bijection of the fold value `x` +/// (`x ^ w1`, `rotl(x, 7) ^ w2`, `x + w0`), so the rewrite is injective in `x` and a uniform `x` gives a uniform +/// word in every position. Checked over 2^16 consecutive `x` for 16 random `w`. +#[test] +fn rewrite_is_a_bijection_of_the_fold_value() { + let mut rng = SplitMix64::new(0x7363_7261_7463_6801); + for _ in 0..16 { + let w = [rng.next() as u32, rng.next() as u32, rng.next() as u32]; + let x0 = rng.next() as u32; + let mut seen = [vec![false; 1 << 16], vec![false; 1 << 16], vec![false; 1 << 16]]; + for i in 0..(1u32 << 16) { + let x = x0.wrapping_add(i); + let out = scratch_rewrite(x, &w); + for j in 0..3 { + // a bijection of x maps 2^16 consecutive x to 2^16 distinct words; the low 16 bits alone are + // distinct for the xor words (x ^ c) and for the add word (x + c), since both act on the low 16 + // bits as bijections of the low 16 bits of x; the rotl word is checked on its rotated-back bits + let key = if j == 1 { out[j].rotate_right(7) & 0xffff } else { out[j] & 0xffff }; + assert!(!seen[j][key as usize], "word {j} repeats inside 2^16 consecutive x"); + seen[j][key as usize] = true; + } + } + } + // The rewrite inverts: from the old content and any ONE written word the fold value is recovered, so a + // rewritten slot carries exactly 32 bits of new state (the point of question 2's arithmetic). + let w = [0x1234_5678, 0x9abc_def0, 0x0fed_cba9]; + let x = 0xdead_beef; + let out = scratch_rewrite(x, &w); + assert_eq!(out[0] ^ w[1], x); + assert_eq!((out[1] ^ w[2]).rotate_right(7), x); + assert_eq!(out[2].wrapping_sub(w[0]), x); +} + +/// For a fixed (seed, slot, j) the fill is a bijection of the lane nonce: `splitmix32` is a bijection of its +/// 32-bit input and the input `((base + lane) ^ s) + c` is a bijection of `base + lane`. Over 2^16 consecutive +/// nonces no fill word repeats, for 8 slots x 3 words. +#[test] +fn fill_is_a_bijection_of_the_nonce() { + let seed = seed_words_from_bytes(b"igneum-genesis"); + for slot in [0u32, 1, 63, 64, 255, 1023, 2047] { + for j in 0..3u32 { + let mut words: Vec = (0..(1u32 << 16)).map(|n| scratch_fill(&seed, n, 0, slot, j)).collect(); + words.sort_unstable(); + words.dedup(); + assert_eq!(words.len(), 1 << 16, "slot {slot} word {j}: fill words of 2^16 consecutive nonces are distinct"); + } + } + // base + lane is the lane nonce: the fill of lane l at base b is the fill of lane 0 at base b + l + assert_eq!(scratch_fill(&seed, 0x1000, 7, 5, 2), scratch_fill(&seed, 0x1007, 0, 5, 2)); + // and it wraps with the nonce: base 0xffffffe0, lane 31 is nonce 0xffffffff; lane 32 would be nonce 0 + assert_eq!(scratch_fill(&seed, 0xffff_ffe0, 32, 5, 2), scratch_fill(&seed, 0, 0, 5, 2)); + // the three word positions of one slot and nonce are three different permutation outputs + let f: Vec = (0..3).map(|j| scratch_fill(&seed, 12345, 7, 17, j)).collect(); + assert!(f[0] != f[1] && f[1] != f[2] && f[0] != f[2]); +} + +// --------------------------------------------------------------------------------------------------------------- +// 2. Uniformity of the written words and the slot re-hit rate (questions 1 and 2) +// --------------------------------------------------------------------------------------------------------------- + +/// Birthday arithmetic: the expected number of distinct slots after `n` uniform draws from `s` slots. +fn expected_distinct(s: usize, n: usize) -> f64 { + let s = s as f64; + s * (1.0 - (1.0 - 1.0 / s).powi(n as i32)) +} + +struct ClassStats { + units: usize, + events: usize, + hits: usize, + /// ones count per bit of the written words, 3 x 32 + ones: [[u64; 32]; 3], + /// ones count per bit of written XOR read (the change the rewrite makes to the slot) + delta_ones: [[u64; 32]; 3], + /// re-hit depth histogram: how many earlier RMWs the slot had seen in this unit (0 = first touch) + depth: Vec, + max_depth: usize, + /// how often each slot index was addressed (the slot comes from a register's low bits) + slot_hist: Vec, +} + +fn class_stats(name: &str, seeds: &[&str], units_per_seed: usize) -> ClassStats { + let c = class(name); + let mut st = ClassStats { + units: 0, + events: 0, + hits: 0, + ones: [[0; 32]; 3], + delta_ones: [[0; 32]; 3], + depth: vec![0; 256], + max_depth: 0, + slot_hist: vec![0; c.scratch_slots_per_lane()], + }; + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28); + for seed in seeds { + let p = generate_class(seed, c); + assert_eq!(p.scratch_ops_per_hash(), c.scratch_slots() * ITERATIONS); + for u in 0..units_per_seed { + let base = (u as u32).wrapping_mul(32).wrapping_add(0x4000_0000); + let (_, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true); + assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES); + let mut count: HashMap<(u8, u32), usize> = HashMap::new(); + for e in &ev { + assert!(e.slot < c.scratch_slots_per_lane() as u32, "slot inside the lane's scratch"); + let d = count.entry((e.lane, e.slot)).or_insert(0); + assert_eq!(e.hit, *d > 0, "hit flag agrees with the unit's own history"); + assert_eq!(e.written, scratch_rewrite(e.x, &e.read)); + if !e.hit { + let fill = [ + scratch_fill(&p.seed, base, e.lane as u32, e.slot, 0), + scratch_fill(&p.seed, base, e.lane as u32, e.slot, 1), + scratch_fill(&p.seed, base, e.lane as u32, e.slot, 2), + ]; + assert_eq!(e.read, fill, "a first touch reads the fill"); + } + st.depth[(*d).min(255)] += 1; + st.max_depth = st.max_depth.max(*d); + st.slot_hist[e.slot as usize] += 1; + *d += 1; + st.events += 1; + st.hits += e.hit as usize; + for j in 0..3 { + for b in 0..32 { + st.ones[j][b] += ((e.written[j] >> b) & 1) as u64; + st.delta_ones[j][b] += (((e.written[j] ^ e.read[j]) >> b) & 1) as u64; + } + } + } + st.units += 1; + } + } + st +} + +/// Bit bias of every written word (and of the change each rewrite makes) within 6 sigma of a fair coin, over +/// 3 seeds x 2^11 units per class (131,072 hashes per seed set); the slot re-hit rate against the birthday +/// formula within 3 percent relative. The table printed here is the one in the analysis. +#[test] +fn written_words_unbiased_and_rehit_rates() { + let seeds = ["igneum-genesis", "igneum-genesis/stats1", "igneum-genesis/stats2"]; + let units = 1usize << 11; + println!("class | slots/lane | RMW/hash | events | re-hits | re-hit % | birthday % | slot chi2 z (spread) | max depth | max bias sigma | max delta bias sigma"); + for name in CLASSES { + let c = class(name); + let st = class_stats(name, &seeds, units); + let n = st.events as f64; + let sigma = (n / 4.0).sqrt(); + let mut worst = 0.0f64; + let mut worst_delta = 0.0f64; + for j in 0..3 { + for b in 0..32 { + let z = (st.ones[j][b] as f64 - n / 2.0).abs() / sigma; + let zd = (st.delta_ones[j][b] as f64 - n / 2.0).abs() / sigma; + assert!(z <= 6.0, "{name}: written word {j} bit {b} biased: {z:.2} sigma"); + assert!(zd <= 6.0, "{name}: rewrite delta word {j} bit {b} biased: {zd:.2} sigma"); + worst = worst.max(z); + worst_delta = worst_delta.max(zd); + } + } + let per_lane_hash = c.scratch_slots() * ITERATIONS; + let s = c.scratch_slots_per_lane(); + let exp_hits = per_lane_hash as f64 - expected_distinct(s, per_lane_hash); + let exp_pct = 100.0 * exp_hits / per_lane_hash as f64; + let got_pct = 100.0 * st.hits as f64 / st.events as f64; + // chi-square of the slot histogram against uniform (df = s - 1): the slot is a register's low bits, and + // the measured re-hit rate runs above the uniform birthday rate (the finding of the analysis, question 2) + let expect_per_slot = n / s as f64; + let chi2: f64 = st.slot_hist.iter().map(|&h| (h as f64 - expect_per_slot).powi(2) / expect_per_slot).sum(); + let chi2_z = (chi2 - (s as f64 - 1.0)) / (2.0 * (s as f64 - 1.0)).sqrt(); + let hot = *st.slot_hist.iter().max().unwrap() as f64 / expect_per_slot; + let cold = *st.slot_hist.iter().min().unwrap() as f64 / expect_per_slot; + println!( + "{name} | {s} | {per_lane_hash} | {} | {} | {got_pct:.2} | {exp_pct:.2} | {chi2_z:.1} (hottest slot {hot:.2}x, coldest {cold:.2}x) | {} | {worst:.2} | {worst_delta:.2}", + st.events, st.hits, st.max_depth + ); + // a regression band, not a uniformity claim: the rate sits between the uniform birthday rate and twice it + assert!( + got_pct >= 0.9 * exp_pct && got_pct <= 2.0 * exp_pct, + "{name}: re-hit rate {got_pct:.2}% against birthday {exp_pct:.2}%" + ); + // depth histogram: the number of earlier RMWs a re-hit slot had seen in the unit + let shown: Vec = st.depth.iter().take(st.max_depth + 1).enumerate().map(|(d, n)| format!("{d}:{n}")).collect(); + println!(" depth histogram {}", shown.join(" ")); + } +} + +// --------------------------------------------------------------------------------------------------------------- +// 3. Hand-built edge programs against an independent hand model (question 3, CPU half) +// --------------------------------------------------------------------------------------------------------------- + +fn ins(op: Op, dst: u8, src: u8) -> Instr { + Instr { op, dst, src, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 } +} +fn add_imm(dst: u8, src: u8, imm: u32) -> Instr { + Instr { op: Op::Add, dst, src, src2: 0, imm, imm2: imm, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 } +} + +/// A hand-built program of class `c` named `name` (its seed is the name, so its fill words and init words are +/// its own). These bypass the generator and the acceptance rule, like `TESTS.md` section 2; `sub r, r` zeroes a +/// register as the Swift edge set does. +fn edge(name: &str, c: LoadClass, instrs: Vec) -> Program { + let seed_string = format!("igneum-scratch-edge/{name}"); + let seed_bytes = seed_string.as_bytes().to_vec(); + let k = instrs.iter().filter(|i| i.op == Op::Scratch).count(); + assert_eq!(k, c.scratch_slots(), "{name}: the class carries the program's scratch count"); + Program { + seed: seed_words_from_bytes(&seed_bytes), + seed_string, + seed_bytes, + generator: GENERATOR_VERSION, + attempt: 0, + class: c, + era_bytes: None, + instrs, + shadow: Vec::new(), + } +} + +/// The edge set for a scratch of `kb` KiB per warp. Each entry: (name, what it drives, program). +fn edge_programs(kb: u8) -> Vec<(String, &'static str, Program)> { + let m = LoadClass::scratch(1, kb).scratch_slot_mask(); + let dsts = [2u8, 3, 4, 5, 6, 7, 0, 2, 3, 4, 5, 6, 7, 0, 2, 3]; + let scr = |n: usize, src: u8| -> Vec { (0..n).map(|i| ins(Op::Scratch, dsts[i], src)).collect() }; + let mut v = Vec::new(); + // every RMW of the hash to slot 0 through a zero register: 64 dependent RMWs on one slot per lane + let mut p = vec![ins(Op::Sub, 1, 1)]; + p.extend(scr(8, 1)); + v.push(("slot0".to_string(), "r1 = 0: every RMW to slot 0", edge(&format!("slot0/k{kb}"), LoadClass::scratch(8, kb), p))); + // slot MASK through the in-range register MASK + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m)]; + p.extend(scr(8, 1)); + v.push(("slotmask".to_string(), "r1 = MASK: every RMW to the last slot", edge(&format!("slotmask/k{kb}"), LoadClass::scratch(8, kb), p))); + // slot MASK through the out-of-range register 0xffffffff + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, 1), ins(Op::Sub, 1, 2)]; + p.extend(scr(8, 1)); + v.push(("ones".to_string(), "r1 = 0xffffffff: masked to the last slot", edge(&format!("ones/k{kb}"), LoadClass::scratch(8, kb), p))); + // slot 0 through the out-of-range register MASK + 1 + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m.wrapping_add(1))]; + p.extend(scr(8, 1)); + v.push(("maskplus1".to_string(), "r1 = MASK + 1: masked to slot 0", edge(&format!("maskplus1/k{kb}"), LoadClass::scratch(8, kb), p))); + // 16 RMWs per iteration on slot 0: 128 dependent RMWs on one slot per lane per hash + let mut p = vec![ins(Op::Sub, 1, 1)]; + p.extend(scr(16, 1)); + v.push(("sixteen".to_string(), "16 RMWs per iteration on slot 0", edge(&format!("sixteen/k{kb}"), LoadClass::scratch(16, kb), p))); + // one slot per lane from the init words: lanes with equal slots would show any cross-lane aliasing + // (r5 is the slot register and is never a destination here) + let p: Vec = [0u8, 1, 2, 3, 4, 6, 7, 0].iter().map(|&d| ins(Op::Scratch, d, 5)).collect(); + v.push(("lanevar".to_string(), "r5 never written: one init-dependent slot per lane", edge(&format!("lanevar/k{kb}"), LoadClass::scratch(8, kb), p))); + // alternating slot 0 and slot MASK inside one iteration + let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, m)]; + for (i, &d) in [3u8, 4, 5, 6, 7, 0, 3, 4].iter().enumerate() { + // r1 and r2 hold the two slots and are never destinations + p.push(ins(Op::Scratch, d, if i % 2 == 0 { 1 } else { 2 })); + } + v.push(("twoslots".to_string(), "slot 0 and slot MASK alternating", edge(&format!("twoslots/k{kb}"), LoadClass::scratch(8, kb), p))); + v +} + +/// The hand model: a second, minimal interpreter for the ops the edge programs use (sub, add, scratch), with its +/// own slot store keyed by (lane, slot). `mutate` swaps the rewrite's words to show the comparison has teeth. +fn hand_model(p: &Program, base: u32, mutate: bool) -> [u64; 32] { + let seed = &p.seed; + let m = p.class.scratch_slot_mask(); + let mut r = [[0u32; LANES]; 8]; + for lane in 0..LANES { + let nonce = base.wrapping_add(lane as u32); + for i in 0..8 { + let mut x = nonce ^ seed[i]; + x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1)); + x = splitmix32(x); + r[i][lane] = x ^ seed[(i + 1) & 7]; + } + } + let mut store: HashMap<(usize, u32), [u32; 3]> = HashMap::new(); + for _ in 0..ITERATIONS { + let sel = r[0]; + for ins in &p.instrs { + let (d, a) = (ins.dst as usize, ins.src as usize); + match ins.op { + Op::Sub => { + for lane in 0..LANES { + r[d][lane] = r[d][lane].wrapping_sub(r[a][lane]); + } + } + Op::Add => { + for lane in 0..LANES { + let c = if (sel[lane] >> ins.bit) & 1 != 0 { ins.imm2 } else { ins.imm }; + r[d][lane] = r[d][lane].wrapping_add(r[a][lane]).wrapping_add(c); + } + } + Op::Scratch => { + for lane in 0..LANES { + let slot = r[a][lane] & m; + let w = *store.entry((lane, slot)).or_insert_with(|| { + let mut f = [0u32; 3]; + for j in 0..3u32 { + // the fill, written out in full rather than through verify::scratch_fill + let n = base.wrapping_add(lane as u32); + f[j as usize] = splitmix32( + (n ^ seed[j as usize]) + .wrapping_add(slot.wrapping_mul(0x9E37_79B1)) + .wrapping_add((j + 1).wrapping_mul(0x85EB_CA77)), + ); + } + f + }); + let mut x = r[d][lane] ^ w[0]; + x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[1]; + x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[2]; + r[d][lane] = x; + let out = if mutate { + [x.rotate_left(7) ^ w[2], x ^ w[1], x.wrapping_add(w[0])] + } else { + [x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])] + }; + store.insert((lane, slot), out); + } + } + other => panic!("the hand model does not implement {other:?}"), + } + } + } + let mut out = [0u64; 32]; + for lane in 0..LANES { + let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21); + let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27); + out[lane] = ((hi as u64) << 32) | lo as u64; + } + out +} + +/// The four unit bases of every edge vector: 0 and 32 (two consecutive units, the pair a one-warp persistent +/// launch runs on one arena), a unit straddling 2^31, and the unit that wraps past 2^32. +const EDGE_BASES: [u32; 4] = [0, 32, 0x7fff_fff0, 0xffff_ffe0]; + +#[test] +fn edge_programs_match_the_hand_model() { + let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24); + let mut cases = 0; + for kb in [32u8, 128] { + for (name, what, p) in edge_programs(kb) { + let slots = p.class.scratch_slots_per_lane(); + for base in EDGE_BASES { + let (res, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true); + let hand = hand_model(&p, base, false); + assert_eq!(res.hashes, hand, "{name} k{kb} base {base:#x}: interpreter against the hand model ({what})"); + assert_ne!(res.hashes, hand_model(&p, base, true), "{name} k{kb}: the comparison has teeth"); + // the slots the trace saw are the ones the program was built to drive + let slot_set: std::collections::BTreeSet = ev.iter().map(|e| e.slot).collect(); + let m = (slots - 1) as u32; + match name.as_str() { + "slot0" | "maskplus1" | "sixteen" => assert_eq!(slot_set.into_iter().collect::>(), vec![0]), + "slotmask" | "ones" => assert_eq!(slot_set.into_iter().collect::>(), vec![m]), + "twoslots" => assert_eq!(slot_set.into_iter().collect::>(), vec![0, m]), + "lanevar" => { + for e in &ev { + assert!(e.slot <= m); + } + } + _ => unreachable!(), + } + // the chain depth on the driven slot: every RMW after the first per lane is a re-hit + let per_lane = p.scratch_ops_per_hash(); + let hits = ev.iter().filter(|e| e.hit).count(); + let expected_hits = match name.as_str() { + "twoslots" => (per_lane - 2) * LANES, + _ => (per_lane - 1) * LANES, + }; + assert_eq!(hits, expected_hits, "{name} k{kb}: re-hits"); + cases += 1; + } + } + } + assert_eq!(cases, 2 * 7 * 4); +} + +// --------------------------------------------------------------------------------------------------------------- +// 4. The static scratch check over every emitted kernel of every scr pack (question 4) +// --------------------------------------------------------------------------------------------------------------- + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Dialect { + Metal, + Cuda, + OpenCl, +} + +/// The static scratch check: every scratch read-modify-write in an emitted kernel has the one masked form the +/// emitter writes, the arena is the lane's own `slots x 4` words, the tag is `salt + unit`, and nothing else +/// touches the scratch. Like the dataset mask check of `TESTS.md` section 5 and `tests/packs.rs`, a text check: +/// the guarantee is that the emitter has one template and it masks. +pub fn scratch_text_check(text: &str, dialect: Dialect, k: usize, slots: usize, kernels: usize) -> Result<(), String> { + assert!(kernels >= 1); + // every count below is per hash kernel; an OpenCL bound file carries igneum_hash and igneum_hash_bound + let k = k * kernels; + assert!(slots.is_power_of_two() && slots >= 1); + let mask = (slots - 1) as u32; + let wpl = slots * 4; + let (u, load, store, ptr) = match dialect { + Dialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "device uint* arena"), + Dialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "uint32_t* arena"), + Dialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); }", "__global uint* arena"), + }; + let count = |needle: &str| text.matches(needle).count(); + let mut errs = Vec::new(); + let mut expect = |what: &str, got: usize, want: usize| { + if got != want { + errs.push(format!("{what}: {got}, expected {want}")); + } + }; + // k slot computations, each masked with exactly the class's mask and immediately followed by the one load form + expect("slot definitions `{ u s_ = r`", count(&format!("{{ {u} s_ = r")), k); + expect("masked slot followed by the load", count(&format!(" & {mask}u; {load}")), k); + expect("stores of the tagged slot", count(store), k); + expect("tag compares", count("(v_.x == tag)"), k); + expect("fill calls (three per RMW)", count("scr_fill(gbase, lane, s_, "), 3 * k); + // the arena: one definition with the class's words per lane, and 2k uses (one load, one store per RMW) + expect("arena definition", count(&format!("{ptr} = scratch + ((size_t)warp_ * 32u + lane) * {wpl}u;")), kernels); + expect("arena mentions (definition + load + store per RMW)", count("arena"), kernels + 2 * k); + expect("tag definition `tag = salt + g_`", count(&format!("{u} tag = salt + g_;")), kernels); + expect("direct scratch indexing", count("scratch["), 0); + expect("scratch pointer arithmetic outside the arena definition", count("scratch +"), kernels); + // no other mask value on a slot: every `s_ = r` line carries the class mask and nothing else carries ` & Nu; uint4 v_` + let any_mask_load = count(&format!("u; {load}")); + expect("loads preceded by some mask (must all be the class mask)", any_mask_load, k); + if errs.is_empty() { + Ok(()) + } else { + Err(errs.join("; ")) + } +} + +fn packs_rw_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-readwidth") +} + +fn scr_packs() -> Vec { + let mut v: Vec = std::fs::read_dir(packs_rw_dir()) + .unwrap() + .map(|d| d.unwrap().file_name().to_string_lossy().to_string()) + .filter(|n| n.starts_with("scr")) + .collect(); + v.sort(); + v +} + +fn read_pack(pack: &str, file: &str) -> String { + let p = packs_rw_dir().join(pack).join(file); + std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display())) +} + +/// Every scr pack regenerates from its program.json (seed bytes, class, day bytes, size) to the same six kernel +/// texts, byte for byte, and every one of those texts passes the static scratch check for the class's k and slot +/// count; the check fails on four deliberate breaks of a copy of the Metal text (mask dropped, mask changed, arena +/// stride changed, a stray scratch access) and on the OpenCL and CUDA twins of the first. +#[test] +fn scr_packs_regenerate_and_pass_the_static_scratch_check() { + let packs = scr_packs(); + assert!(packs.len() >= 6, "the scr packs: {packs:?}"); + let mut checked = 0; + let mut sample_metal = String::new(); + let mut sample_cl = String::new(); + let mut sample_cu = String::new(); + let mut sample_k = 0; + let mut sample_slots = 0; + for pack in &packs { + let j: Value = serde_json::from_str(&read_pack(pack, "program.json")).unwrap(); + let name = j["load_class"].as_str().unwrap(); + let c = class(name); + assert_eq!(&format!("{name}"), pack, "pack directory named after its class"); + let seed = j["seed"].as_str().unwrap(); + let seed_bytes = igneum_pow::bind::unhex(j["seed_bytes"].as_str().unwrap()).unwrap(); + let day_bytes = igneum_pow::bind::unhex(j["dataset"]["day_bytes"].as_str().unwrap()).unwrap(); + let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32; + assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard"); + let program = generate_from_seed_bytes_class(seed, &seed_bytes, c); + assert_eq!(program.class, c); + assert_eq!(program.program_id(), u64::from_str_radix(j["program_id"].as_str().unwrap().trim_start_matches("0x"), 16).unwrap()); + let mut dataset = DatasetSource::from_key(seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2); + dataset.key_bytes = day_bytes; + let e = Epoch { program, dataset }; + let p = &e.program; + let mp = e.dataset.memhard().map(|m| &m.params); + let k = c.scratch_slots(); + let slots = c.scratch_slots_per_lane(); + assert_eq!(p.scratch_ops_per_hash(), k * ITERATIONS); + for (file, text, dialect, kernels) in [ + ("program.metal", metal_program(p, log2, LoadSource::Stored), Dialect::Metal, 1), + ("program_bound.metal", metal_program_bound(p, log2), Dialect::Metal, 1), + ("kernel.cu", cuda_kernel(p, mp), Dialect::Cuda, 1), + ("kernel_bound.cu", cuda_kernel_bound(p, mp), Dialect::Cuda, 1), + ("kernel.cl", opencl_kernel(p, mp), Dialect::OpenCl, 1), + // the OpenCL bound file carries igneum_hash and igneum_hash_bound + ("kernel_bound.cl", opencl_kernel_bound(p, mp), Dialect::OpenCl, 2), + ] { + let on_disk = read_pack(pack, file); + assert_eq!(on_disk, text, "{pack}/{file}: the pack is the emitter's text"); + // scr0 is the persistent control: an arena and a tag, no read-modify-write; the check holds with k = 0 + scratch_text_check(&on_disk, dialect, k, slots, kernels).unwrap_or_else(|e| panic!("{pack}/{file}: {e}")); + checked += 1; + } + // the vectors of the pack are the CPU's + let v: Value = serde_json::from_str(&read_pack(pack, "vectors.json")).unwrap(); + for w in v["warps"].as_array().unwrap() { + let base = w["base_nonce"].as_u64().unwrap() as u32; + let got = e.hash_warp(base); + for (lane, x) in w["expected"].as_array().unwrap().iter().enumerate() { + let want = u64::from_str_radix(x.as_str().unwrap().trim_start_matches("0x"), 16).unwrap(); + assert_eq!(got[lane], want, "{pack}: base {base} lane {lane}"); + } + } + if k == 4 && slots == 64 { + sample_metal = read_pack(pack, "program.metal"); + sample_cl = read_pack(pack, "kernel.cl"); + sample_cu = read_pack(pack, "kernel.cu"); + sample_k = k; + sample_slots = slots; + } + } + assert_eq!(checked, packs.len() * 6); + println!("static scratch check: {checked} kernels over {} scr packs", packs.len()); + + // The deliberate breaks (the watcher rule of CLAUDE.md: a check is trusted once it fails on a known-broken + // case). Each must be caught; the message names what. + assert!(sample_k == 4 && sample_slots == 64, "scr4k32 is in the pack set"); + let mask = format!(" & {}u; uint4 v_", sample_slots - 1); + let broken_mask = sample_metal.replacen(&mask, "; uint4 v_", 1); + assert_ne!(broken_mask, sample_metal); + let e = scratch_text_check(&broken_mask, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); + println!("break 1 (one mask dropped, Metal): {e}"); + let wrong_mask = sample_metal.replace(" & 63u;", " & 127u;"); + let e = scratch_text_check(&wrong_mask, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 0, expected 4"), "{e}"); + println!("break 2 (mask 63 -> 127 on every RMW, Metal): {e}"); + let wrong_stride = sample_metal.replace("* 256u;", "* 128u;"); + let e = scratch_text_check(&wrong_stride, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("arena definition: 0, expected 1"), "{e}"); + println!("break 3 (arena stride 256 -> 128 words, Metal): {e}"); + let stray = format!("{sample_metal}\n// stray\n// arena[0] = 0u; scratch[1] = 1u;\n"); + let e = scratch_text_check(&stray, Dialect::Metal, 4, 64, 1).unwrap_err(); + assert!(e.contains("arena mentions") && e.contains("direct scratch indexing: 1, expected 0"), "{e}"); + println!("break 4 (a stray arena and scratch access, Metal): {e}"); + let e = scratch_text_check(&sample_cl.replacen(" & 63u; uint4 v_ = vload4", "; uint4 v_ = vload4", 1), Dialect::OpenCl, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); + println!("break 5 (one mask dropped, OpenCL): {e}"); + let e = scratch_text_check(&sample_cu.replacen(" & 63u; uint4 v_ = *(const uint4*)", "; uint4 v_ = *(const uint4*)", 1), Dialect::Cuda, 4, 64, 1).unwrap_err(); + assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}"); + println!("break 6 (one mask dropped, CUDA): {e}"); + // and the unbroken texts pass under the same calls + scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 1).unwrap(); + scratch_text_check(&sample_cl, Dialect::OpenCl, 4, 64, 1).unwrap(); + scratch_text_check(&sample_cu, Dialect::Cuda, 4, 64, 1).unwrap(); + // a wrong slot count, RMW count or kernel count against a right text fails too (the check is tied to the class) + assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 256, 1).is_err()); + assert!(scratch_text_check(&sample_metal, Dialect::Metal, 3, 64, 1).is_err()); + assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 2).is_err()); +} + +// --------------------------------------------------------------------------------------------------------------- +// 5. The fuzz: 200 generated scratch programs, contract on every instruction, 4 units each across the 32-bit +// range including the wrap; with IGNEUM_SCRATCH_PACKS_OUT the packs for the Metal runs (question 3, 4) +// --------------------------------------------------------------------------------------------------------------- + +/// Write a pack whose vectors.json carries `bases` (any number of units) instead of the three standard bases. +fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) -> Vec<[u64; 32]> { + let mut pack = export_pack(e, day, source); + let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect(); + let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true); + for f in pack.files.iter_mut() { + if f.0 == "vectors.json" { + f.1 = vj.clone(); + } + } + pack.write_to(dir).unwrap(); + outs +} + +fn contract(p: &Program) { + assert_eq!(p.instrs.len(), INSTR_COUNT); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count() + p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), 16); + assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), p.class.scratch_slots()); + assert!(p.instrs[0].op != Op::Load && p.instrs[0].op != Op::Scratch, "instruction 0 is never a memory op"); + for (k, i) in p.instrs.iter().enumerate() { + assert!(i.src != i.dst, "#{k}: src == dst"); + assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot); + assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask); + assert!(i.dst < 8 && i.src < 8 && i.src2 < 8); + assert_eq!(i.width, 1, "#{k}: a scratch class reads one-word loads"); + } + assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program"); +} + +#[test] +fn fuzz_scr_programs_cpu() { + let n: usize = std::env::var("IGNEUM_SCRATCH_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200); + // IGNEUM_FUZZ_SEED_BASE (default 0): the seed index offset for the continuous fuzzer (infra/build-server/capacity) + let base: usize = std::env::var("IGNEUM_FUZZ_SEED_BASE").ok().and_then(|s| s.parse().ok()).unwrap_or(0); + let out = std::env::var("IGNEUM_SCRATCH_PACKS_OUT").ok().map(PathBuf::from); + let mut rng = SplitMix64::new(0x6967_6e65_756d_2d73); // "igneum-s" + let day = "2026-10-03"; + let closed = DatasetSource::new(day, DatasetMode::ClosedForm, 28); + // memory-hard sources per size, built once each (the cache fill is 0.2 s); only when packs are written + let mut mh: HashMap = HashMap::new(); + let mut manifest = String::from("pack\tclass\tlog2\tprogram_id\tscratch_ops_per_hash\tbases\n"); + let mut per_class: HashMap = HashMap::new(); + let mut units = 0usize; + let mut wraps = 0usize; + if let Some(dir) = &out { + std::fs::create_dir_all(dir).unwrap(); + // the edge packs first: 64 MiB datasets (no dataset load in them), the four edge bases + for kb in [32u8, 128] { + for (name, _what, p) in edge_programs(kb) { + let log2 = 24; + let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2)); + let e = Epoch { program: p, dataset: ds }; + let pack_name = format!("edge-{name}-k{kb}"); + write_pack_with_bases(&dir.join(&pack_name), &e, day, &EDGE_BASES, "igneum-pow tests/scratch.rs edge"); + manifest.push_str(&format!( + "{pack_name}\t{}\t{log2}\t{:016x}\t{}\t{}\n", + e.program.class.name(), + e.program.program_id(), + e.program.scratch_ops_per_hash(), + EDGE_BASES.iter().map(|b| format!("{b}")).collect::>().join(",") + )); + mh.insert(log2, e.dataset); + } + } + } + for i in base..base + n { + let name = CLASSES[rng.below(CLASSES.len() as u64) as usize]; + let c = class(name); + let seed = format!("igneum-scratch-fuzz/{i}"); + let p = generate_class(&seed, c); + contract(&p); + *per_class.entry(name.to_string()).or_insert(0) += 1; + // four bases: one inside a 256-nonce batch (in-batch check on the GPU), one straddling 2^31, one in + // the last 256 nonces (the unit wraps past 2^32 or ends on it), one uniform + let b0 = (rng.below(8) as u32) * 32; + let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32); + let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32); + let b3 = (rng.next() as u32) & !31; + let bases = [b0, b1, b2, b3]; + // an aligned unit never straddles 2^32 (spec 1.9); the top unit ends on 0xffffffff and the persistent + // kernel's unit sequence wraps inside a launch, which the Metal run checks with packbench --batch-base + wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count(); + // the CPU: the interpreter is deterministic and every scratch event is inside the lane's slots + for &b in &bases { + let (r1, ev) = interpret_warp_scratch(&p, &p.seed, b, &closed, true); + let r2 = interpret_warp_scratch(&p, &p.seed, b, &closed, false).0; + assert_eq!(r1.hashes, r2.hashes); + assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES); + assert!(ev.iter().all(|e: &ScratchEvent| e.slot < c.scratch_slots_per_lane() as u32)); + units += 1; + } + if let Some(dir) = &out { + let log2 = [24u32, 26, 28][rng.below(3) as usize]; + let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2)); + let e = Epoch { program: p, dataset: ds }; + let pack_name = format!("fuzz-{i:03}-{name}-l{log2}"); + write_pack_with_bases(&dir.join(&pack_name), &e, day, &bases, "igneum-pow tests/scratch.rs fuzz"); + manifest.push_str(&format!( + "{pack_name}\t{name}\t{log2}\t{:016x}\t{}\t{}\n", + e.program.program_id(), + e.program.scratch_ops_per_hash(), + bases.iter().map(|b| format!("{b}")).collect::>().join(",") + )); + mh.insert(log2, e.dataset); + } else { + let _ = rng.below(3); + } + } + let mut classes: Vec<_> = per_class.iter().collect(); + classes.sort(); + println!("fuzz: {n} programs, {units} units on the CPU, {wraps} units in the top 256 nonces, classes {classes:?}, seeds {base}..{}", base + n); + assert_eq!(units, 4 * n); + assert_eq!(wraps, n, "every program has a unit in the top 256 nonces"); + if let Some(dir) = &out { + std::fs::write(dir.join("manifest.tsv"), manifest).unwrap(); + println!("packs written to {}", dir.display()); + } +} + +/// The fold and rewrite, restated: a slot after `d` dependent RMWs holds 96 bits that are a function of the fill +/// (3 words, a pure function of nonce, slot and seed) and the `d` fold values; a chip that keeps the `d` fold +/// values (32 bits each) instead of the 96-bit slot recomputes the slot in `d` rewrites. This test pins the +/// arithmetic the analysis uses (question 2): the replay from the fold values reproduces the slot. +#[test] +fn slot_is_replayable_from_its_fold_values() { + let seed = seed_words_from_bytes(b"igneum-genesis"); + let (base, lane, slot) = (0x1234_5600u32, 5u32, 17u32); + let fill = [scratch_fill(&seed, base, lane, slot, 0), scratch_fill(&seed, base, lane, slot, 1), scratch_fill(&seed, base, lane, slot, 2)]; + let mut rng = SplitMix64::new(99); + let dsts: Vec = (0..64).map(|_| rng.next() as u32).collect(); + // the honest sequence: read, fold, rewrite, 64 times + let mut w = fill; + let mut xs = Vec::new(); + for &d in &dsts { + let x = fold_words(d, &w); + xs.push(x); + w = scratch_rewrite(x, &w); + } + // the replay: from the fill and the stored fold values alone + let mut w2 = fill; + for &x in &xs { + w2 = scratch_rewrite(x, &w2); + } + assert_eq!(w, w2); + // and nothing shorter: the fold value at step d depends on the slot content at step d, which depends on + // every earlier fold value (drop one and the chain diverges) + let mut w3 = fill; + for (i, &x) in xs.iter().enumerate() { + if i != 10 { + w3 = scratch_rewrite(x, &w3); + } + } + assert_ne!(w, w3); +} diff --git a/tools/attack/adv-accept-v5/inputs/v5-dn3-epoch0-leaves.bin b/tools/attack/adv-accept-v5/inputs/v5-dn3-epoch0-leaves.bin new file mode 100644 index 000000000..15a58456f Binary files /dev/null and b/tools/attack/adv-accept-v5/inputs/v5-dn3-epoch0-leaves.bin differ diff --git a/tools/attack/adv-accept-v5/inputs/v5-dn3-epoch0-program.json b/tools/attack/adv-accept-v5/inputs/v5-dn3-epoch0-program.json new file mode 100644 index 000000000..571230450 --- /dev/null +++ b/tools/attack/adv-accept-v5/inputs/v5-dn3-epoch0-program.json @@ -0,0 +1,416 @@ +{ + "format": "igneum-program-pack-3", + "generator": 5, + "attempt": 0, + "program_id": "0xe5a4ac5978462156", + "program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32", + "dataset_mode": "memory-hard", + "seed": "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000", + "seed_bytes": "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925", + "seed_words": ["0xbe8c5a0c", "0x7646e626", "0x508e29d0", "0xfb8a5b4a", "0x53c50955", "0x685d62fc", "0x6065c013", "0x881096eb"], + "seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32", + "generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried", + "lanes": 32, + "registers": 8, + "iterations": 8, + "instruction_count": 64, + "loads_per_hash": 128, + "program_class": "v5", + "era_seed_bytes": "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925", + "state": { + "block": "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925", + "block_number": 0, + "root": "7e37a9fb19b154d32daf5bf30a50d339a75029fbc9eec9ea20e95439dba5a311", + "leaves": 11, + "records": 11, + "sampled": false, + "leaves_fnv1a64": "0xf141bfee2a8b11b0", + "leaf_derivation": "leaves[i] = Blake2b-512('igneum-sd1/' || root || i_le32 || record_i) as 16 little-endian words; item t XORs leaves[t mod leaves] into its 16 initial words before the first mixer", + "file": "leaves.bin" + }, + "load_class": "mx8-eraaf3a9139+sh256x27+state", + "mixer_mult": 8, + "cache_growth": true, + "mixer": "class v3 (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): every mixer application of the item derivation is 8 applications with round keys (r * 8 + j + 1) * 0x9E3779B9, the 8 dependent cache reads per item unchanged; cache growth rule option C: cache words = 2^(26 + doublings(day)), dataset words = 2^(genesis_log2 + doublings(day)), doublings(day) = floor(log2(1 + day / 1460)) for day = days since genesis", + "load_slots": 16, + "load_mix_percent_4_16_64": [100, 0, 0], + "load_width_counts_4_16_64": [16, 0, 0], + "bytes_per_hash": 512, + "wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots", + "era": { + "label": "af3a9139", + "seed_words": ["0xaf3a9139", "0x894db311", "0xd6173573", "0xc346bc0f", "0x2d547cec", "0xd5634ab8", "0xb9584827", "0x305d030e"], + "draw": "docs/plans/era-layout.md 1.1: SplitMix64 seeded with seed_words[0] | seed_words[1] << 32 of seed_words_from_bytes('igneum-era/' || n_le64 || E_n); width = allowed[below(|allowed|)], stride_mul = low32(next()) | 1, stride_rot = 1 + below(31), then four next() draws for a partial Fisher-Yates over positions log2(W)..15 of which 4 - log2(W) are used", + "allowed_widths": [1], + "width_words": 1, + "stride_mul": "0x2cb18b35", + "stride_rot": 9, + "interleave": [1, 9, 13, 14], + "address": "y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words", + "windows": "per instruction, after the width roll: win = below(3), off = low32(next()) & (2^win - 1); used on a load slot (the instruction's win and off fields)", + "dataset_word": "dataset[w] = item(t(w))[j(w)]: j(w) gathers the bits of w at the interleave positions, t(w) is w with those bits removed", + "program_id_suffix": "'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]" + }, + "op_mix": {"load": 16, "add": 11, "shfl": 7, "mad": 6, "xor": 6, "mul": 4, "mulhi": 4, "rotr": 4, "rotl": 3, "sub": 2, "or": 1}, + "register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]", + "splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16", + "iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order", + "output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo", + "op_semantics": { + "add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)", + "sub": "dst = dst - src", + "mul": "dst = dst * src (low 32)", + "mulhi": "dst = high 32 bits of dst * src", + "xor": "dst = dst ^ src", + "or": "dst = dst | src", + "rotl": "dst = rotl(dst, rot), rot in 1..31", + "rotr": "dst = rotr(dst, src & 31)", + "mad": "dst = src * src2 + dst", + "shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp", + "load": "dst = dst ^ dataset[src & dataset.mask]", + "wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)" + }, + "dataset": { + "log2_words": 28, + "bytes": 1073741824, + "mask": "0x0fffffff", + "day": "bytes:69676e65756d2d6461792ffd50000000000000", + "day_bytes": "69676e65756d2d6461792ffd50000000000000", + "day_words_from": "seed_words_from_bytes(day_bytes)", + "d0": "0xe23f9008", + "d1": "0xde33e763", + "mode": "memory-hard", + "spec": "proto-metal/MEMHARD.md", + "key": ["0xe23f9008", "0xde33e763", "0xc5ba415d", "0x8ddf6786", "0x59f4a4be", "0x8a3bc680", "0x701b8e40", "0x3b59025a"], + "key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]", + "cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"}, + "mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [28, 11, 18, 18, 13, 28, 5, 13], "mul": ["0xd1d79e7f", "0xbcb61f8b", "0x970e91ab", "0xd749d96d", "0x143d7339", "0x2cde0d69", "0x12d0a8c1", "0x2d335be9", "0x80a8aae9", "0x7ac896c7", "0x9de23db7", "0xc362827d", "0x5f4cdb5b", "0xfc6c5097", "0x6f547f83", "0x1ac31b47"], "rc": ["0xe5bef5a3", "0xdb2a7d90", "0xcd1fa7e1", "0x30289419", "0x0a730d58", "0x432f8579", "0x6ab978a5", "0x8c984f49", "0x788c3c9e", "0x051eef02", "0x05e77db9", "0x8b3cd4f3", "0x6948d6cf", "0x0d8677f6", "0xf9505c8c", "0x4513c163"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"}, + "mixer_mult": 8, + "item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: for j in 0..7: s = M(s, rk = (r * 8 + j + 1) * 0x9E3779B9); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then for j in 0..7: s = M(s, rk = (64 + j + 1) * 0x9E3779B9); item(t) = s", + "word": "dataset[w] = item(w >> 4)[w & 15]" + }, + "shadow": {"instrs": 256, "reps": 27, "instrs_per_hash": 55296, "op_mix": {"add": 45, "mad": 31, "shfl": 31, "xor": 30, "rotl": 28, "mul": 22, "or": 18, "mulhi": 17, "rotr": 17, "sub": 17}, "rule": "Counter ASIC 3.0 item 8 (docs/analysis/latency-shadow-2026-10-06.md): after the 64 base instructions the program stream draws instrs more ALU instructions (the non-load table, nine draws each, the source as on an ALU slot); the block runs reps times at the end of every iteration with the iteration's sel; the acceptance rule interprets the base program only", "program_id_suffix": "'shadow/' || instrs_le16 || reps_le16", "instructions": [ + {"i": 0, "op": "or", "dst": 6, "src": 0, "src2": 2, "imm": "0xc1eb1523", "imm2": "0x925958ea", "rot": 14, "bit": 14, "mask": 8}, + {"i": 1, "op": "mul", "dst": 7, "src": 3, "src2": 4, "imm": "0xc7044334", "imm2": "0x755566b6", "rot": 19, "bit": 17, "mask": 8}, + {"i": 2, "op": "mad", "dst": 3, "src": 5, "src2": 5, "imm": "0x33b52f17", "imm2": "0xcc459a7a", "rot": 5, "bit": 19, "mask": 16}, + {"i": 3, "op": "add", "dst": 7, "src": 4, "src2": 4, "imm": "0x97d3d105", "imm2": "0x38e58a06", "rot": 11, "bit": 6, "mask": 1}, + {"i": 4, "op": "add", "dst": 6, "src": 0, "src2": 6, "imm": "0x29b87b9a", "imm2": "0x9629673d", "rot": 16, "bit": 20, "mask": 8}, + {"i": 5, "op": "shfl", "dst": 6, "src": 2, "src2": 5, "imm": "0xb61f2e77", "imm2": "0x30e3807c", "rot": 1, "bit": 7, "mask": 1}, + {"i": 6, "op": "mad", "dst": 5, "src": 3, "src2": 5, "imm": "0x0de66253", "imm2": "0xfdfdd03e", "rot": 4, "bit": 27, "mask": 4}, + {"i": 7, "op": "rotl", "dst": 3, "src": 7, "src2": 1, "imm": "0x688514dd", "imm2": "0x139d08b5", "rot": 5, "bit": 1, "mask": 1}, + {"i": 8, "op": "add", "dst": 0, "src": 7, "src2": 3, "imm": "0xa7c1fb0f", "imm2": "0x4a5f55e1", "rot": 15, "bit": 12, "mask": 1}, + {"i": 9, "op": "mul", "dst": 5, "src": 7, "src2": 2, "imm": "0x1a64de5a", "imm2": "0xe37b0875", "rot": 5, "bit": 27, "mask": 16}, + {"i": 10, "op": "xor", "dst": 4, "src": 5, "src2": 3, "imm": "0x750d4f1a", "imm2": "0x2f3d9402", "rot": 3, "bit": 14, "mask": 2}, + {"i": 11, "op": "mulhi", "dst": 6, "src": 4, "src2": 5, "imm": "0x2eaa86c6", "imm2": "0x9fd5b22f", "rot": 30, "bit": 10, "mask": 1}, + {"i": 12, "op": "rotl", "dst": 5, "src": 2, "src2": 4, "imm": "0x8df50473", "imm2": "0x23d780ce", "rot": 13, "bit": 9, "mask": 1}, + {"i": 13, "op": "mad", "dst": 0, "src": 2, "src2": 1, "imm": "0xa956a7e6", "imm2": "0x58dbb356", "rot": 31, "bit": 11, "mask": 8}, + {"i": 14, "op": "mulhi", "dst": 2, "src": 4, "src2": 6, "imm": "0x6fbfb145", "imm2": "0x064eb015", "rot": 20, "bit": 2, "mask": 4}, + {"i": 15, "op": "or", "dst": 3, "src": 1, "src2": 5, "imm": "0xd313bc27", "imm2": "0x9f0fc235", "rot": 17, "bit": 16, "mask": 1}, + {"i": 16, "op": "sub", "dst": 0, "src": 5, "src2": 0, "imm": "0x86cbd336", "imm2": "0x94715167", "rot": 17, "bit": 31, "mask": 8}, + {"i": 17, "op": "mul", "dst": 0, "src": 6, "src2": 3, "imm": "0x95911c6b", "imm2": "0x816cea32", "rot": 23, "bit": 1, "mask": 2}, + {"i": 18, "op": "shfl", "dst": 5, "src": 1, "src2": 3, "imm": "0xf563dcad", "imm2": "0xb2889856", "rot": 14, "bit": 24, "mask": 4}, + {"i": 19, "op": "or", "dst": 3, "src": 4, "src2": 2, "imm": "0x7bb48d19", "imm2": "0xbf95ad6f", "rot": 8, "bit": 30, "mask": 16}, + {"i": 20, "op": "mad", "dst": 6, "src": 3, "src2": 2, "imm": "0xeb686a8d", "imm2": "0x8bda9f8a", "rot": 11, "bit": 18, "mask": 1}, + {"i": 21, "op": "rotl", "dst": 0, "src": 6, "src2": 0, "imm": "0x0915df51", "imm2": "0xa1365cda", "rot": 11, "bit": 25, "mask": 1}, + {"i": 22, "op": "rotl", "dst": 3, "src": 0, "src2": 3, "imm": "0xc09e4cdc", "imm2": "0x50c1369d", "rot": 25, "bit": 31, "mask": 2}, + {"i": 23, "op": "xor", "dst": 6, "src": 4, "src2": 1, "imm": "0xce01abec", "imm2": "0x5ff8c52e", "rot": 30, "bit": 19, "mask": 8}, + {"i": 24, "op": "or", "dst": 4, "src": 2, "src2": 4, "imm": "0xdacf9e4b", "imm2": "0x086a5684", "rot": 2, "bit": 10, "mask": 2}, + {"i": 25, "op": "add", "dst": 1, "src": 6, "src2": 3, "imm": "0x53e74799", "imm2": "0xfc2f6d17", "rot": 8, "bit": 7, "mask": 16}, + {"i": 26, "op": "rotr", "dst": 7, "src": 6, "src2": 7, "imm": "0xc89a75a0", "imm2": "0x67813253", "rot": 24, "bit": 13, "mask": 1}, + {"i": 27, "op": "add", "dst": 7, "src": 5, "src2": 4, "imm": "0xb8fae90e", "imm2": "0x787048dc", "rot": 26, "bit": 27, "mask": 1}, + {"i": 28, "op": "xor", "dst": 4, "src": 1, "src2": 0, "imm": "0x4c90c2ea", "imm2": "0x2ce7865b", "rot": 4, "bit": 7, "mask": 2}, + {"i": 29, "op": "xor", "dst": 6, "src": 1, "src2": 6, "imm": "0x6131ab08", "imm2": "0xd1825676", "rot": 4, "bit": 30, "mask": 2}, + {"i": 30, "op": "mulhi", "dst": 2, "src": 0, "src2": 1, "imm": "0xbe4a545f", "imm2": "0xdc29e823", "rot": 17, "bit": 5, "mask": 8}, + {"i": 31, "op": "rotr", "dst": 1, "src": 4, "src2": 1, "imm": "0x4d2f9cdc", "imm2": "0xdee1cc73", "rot": 5, "bit": 16, "mask": 4}, + {"i": 32, "op": "xor", "dst": 7, "src": 4, "src2": 5, "imm": "0x9d277aaa", "imm2": "0xa902fca5", "rot": 11, "bit": 3, "mask": 1}, + {"i": 33, "op": "mad", "dst": 3, "src": 2, "src2": 6, "imm": "0x7db09644", "imm2": "0x077bf3d9", "rot": 30, "bit": 19, "mask": 2}, + {"i": 34, "op": "add", "dst": 3, "src": 1, "src2": 1, "imm": "0xd2dc42eb", "imm2": "0x9ffd510b", "rot": 17, "bit": 12, "mask": 2}, + {"i": 35, "op": "add", "dst": 4, "src": 7, "src2": 0, "imm": "0xad8eda6f", "imm2": "0xf871ba37", "rot": 24, "bit": 7, "mask": 4}, + {"i": 36, "op": "add", "dst": 7, "src": 4, "src2": 3, "imm": "0xb0cef8f4", "imm2": "0xd02d30ca", "rot": 22, "bit": 3, "mask": 16}, + {"i": 37, "op": "mad", "dst": 4, "src": 3, "src2": 2, "imm": "0x6ed99a66", "imm2": "0x83dc97c0", "rot": 4, "bit": 1, "mask": 4}, + {"i": 38, "op": "mad", "dst": 5, "src": 7, "src2": 7, "imm": "0xe24fdd24", "imm2": "0x360f5c8b", "rot": 19, "bit": 9, "mask": 8}, + {"i": 39, "op": "add", "dst": 3, "src": 6, "src2": 4, "imm": "0x2bd506e2", "imm2": "0x82e98a22", "rot": 28, "bit": 26, "mask": 1}, + {"i": 40, "op": "rotl", "dst": 0, "src": 2, "src2": 1, "imm": "0xdb1faed3", "imm2": "0xf968a20d", "rot": 21, "bit": 30, "mask": 16}, + {"i": 41, "op": "or", "dst": 1, "src": 0, "src2": 1, "imm": "0xe6dfe522", "imm2": "0xb75813da", "rot": 10, "bit": 13, "mask": 1}, + {"i": 42, "op": "add", "dst": 7, "src": 5, "src2": 7, "imm": "0xc0386e49", "imm2": "0x6435f6d8", "rot": 26, "bit": 28, "mask": 16}, + {"i": 43, "op": "mulhi", "dst": 1, "src": 0, "src2": 4, "imm": "0xda497f7d", "imm2": "0xc4b748f4", "rot": 30, "bit": 10, "mask": 4}, + {"i": 44, "op": "add", "dst": 5, "src": 6, "src2": 0, "imm": "0x3a8921a4", "imm2": "0x1f682c68", "rot": 12, "bit": 12, "mask": 1}, + {"i": 45, "op": "add", "dst": 4, "src": 5, "src2": 6, "imm": "0xeeca4334", "imm2": "0xf246e180", "rot": 26, "bit": 1, "mask": 1}, + {"i": 46, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0x052178c3", "imm2": "0xaab31f10", "rot": 28, "bit": 16, "mask": 8}, + {"i": 47, "op": "xor", "dst": 2, "src": 5, "src2": 0, "imm": "0x91fd7273", "imm2": "0xacc42b07", "rot": 22, "bit": 27, "mask": 4}, + {"i": 48, "op": "mad", "dst": 7, "src": 4, "src2": 1, "imm": "0x05a2970f", "imm2": "0x14876824", "rot": 10, "bit": 5, "mask": 1}, + {"i": 49, "op": "mad", "dst": 2, "src": 6, "src2": 1, "imm": "0x4440d6de", "imm2": "0x352023f9", "rot": 10, "bit": 25, "mask": 8}, + {"i": 50, "op": "mad", "dst": 7, "src": 0, "src2": 7, "imm": "0xa6f82633", "imm2": "0x78465772", "rot": 13, "bit": 8, "mask": 4}, + {"i": 51, "op": "add", "dst": 5, "src": 6, "src2": 5, "imm": "0x745776d8", "imm2": "0x41bd68cc", "rot": 29, "bit": 26, "mask": 1}, + {"i": 52, "op": "shfl", "dst": 3, "src": 0, "src2": 6, "imm": "0x8f94ebac", "imm2": "0xe4a208a0", "rot": 5, "bit": 16, "mask": 1}, + {"i": 53, "op": "or", "dst": 4, "src": 0, "src2": 5, "imm": "0x7b1235f3", "imm2": "0x408b4ea9", "rot": 27, "bit": 19, "mask": 8}, + {"i": 54, "op": "mad", "dst": 0, "src": 2, "src2": 4, "imm": "0x30114050", "imm2": "0xe71dff8f", "rot": 6, "bit": 30, "mask": 1}, + {"i": 55, "op": "rotl", "dst": 5, "src": 1, "src2": 7, "imm": "0x69f92198", "imm2": "0x65496f29", "rot": 9, "bit": 20, "mask": 4}, + {"i": 56, "op": "mul", "dst": 5, "src": 7, "src2": 6, "imm": "0x88b884d4", "imm2": "0x041b67f7", "rot": 25, "bit": 8, "mask": 16}, + {"i": 57, "op": "rotl", "dst": 7, "src": 5, "src2": 6, "imm": "0x7d1e18bc", "imm2": "0xf54ade27", "rot": 27, "bit": 20, "mask": 1}, + {"i": 58, "op": "sub", "dst": 7, "src": 6, "src2": 4, "imm": "0xa457bbab", "imm2": "0x4b05eb29", "rot": 6, "bit": 6, "mask": 4}, + {"i": 59, "op": "rotl", "dst": 2, "src": 7, "src2": 7, "imm": "0x4890bdda", "imm2": "0xadcb3d99", "rot": 22, "bit": 26, "mask": 16}, + {"i": 60, "op": "shfl", "dst": 7, "src": 1, "src2": 7, "imm": "0xe4da58eb", "imm2": "0x1d50080d", "rot": 22, "bit": 26, "mask": 2}, + {"i": 61, "op": "add", "dst": 4, "src": 0, "src2": 5, "imm": "0x60d0ff55", "imm2": "0xcba6a161", "rot": 6, "bit": 7, "mask": 8}, + {"i": 62, "op": "mul", "dst": 1, "src": 5, "src2": 1, "imm": "0x0fdc702e", "imm2": "0x6b1bd03a", "rot": 9, "bit": 23, "mask": 4}, + {"i": 63, "op": "mulhi", "dst": 7, "src": 6, "src2": 1, "imm": "0xcc44ed62", "imm2": "0x931dd006", "rot": 1, "bit": 0, "mask": 2}, + {"i": 64, "op": "mul", "dst": 5, "src": 0, "src2": 7, "imm": "0x448c3aba", "imm2": "0x1136818a", "rot": 23, "bit": 7, "mask": 1}, + {"i": 65, "op": "rotr", "dst": 2, "src": 5, "src2": 1, "imm": "0xb520dfb8", "imm2": "0xdce7d8f4", "rot": 22, "bit": 10, "mask": 8}, + {"i": 66, "op": "mulhi", "dst": 0, "src": 3, "src2": 4, "imm": "0x19fada48", "imm2": "0x3cc04e0e", "rot": 1, "bit": 24, "mask": 16}, + {"i": 67, "op": "or", "dst": 0, "src": 7, "src2": 3, "imm": "0x1c5905b4", "imm2": "0xf896dfb7", "rot": 31, "bit": 30, "mask": 8}, + {"i": 68, "op": "rotl", "dst": 0, "src": 1, "src2": 5, "imm": "0xa28511c1", "imm2": "0xbf649695", "rot": 19, "bit": 27, "mask": 1}, + {"i": 69, "op": "rotl", "dst": 0, "src": 7, "src2": 5, "imm": "0x22e2fef3", "imm2": "0x468bec9e", "rot": 1, "bit": 28, "mask": 1}, + {"i": 70, "op": "or", "dst": 2, "src": 3, "src2": 6, "imm": "0xcb542821", "imm2": "0xe084cf11", "rot": 15, "bit": 5, "mask": 8}, + {"i": 71, "op": "mad", "dst": 3, "src": 7, "src2": 5, "imm": "0xf06b8f6a", "imm2": "0x7382e15f", "rot": 6, "bit": 28, "mask": 16}, + {"i": 72, "op": "mad", "dst": 7, "src": 0, "src2": 4, "imm": "0x4f64c588", "imm2": "0x47b63ee9", "rot": 8, "bit": 10, "mask": 2}, + {"i": 73, "op": "or", "dst": 4, "src": 3, "src2": 2, "imm": "0xff138eea", "imm2": "0x6fd90eda", "rot": 23, "bit": 3, "mask": 16}, + {"i": 74, "op": "xor", "dst": 3, "src": 1, "src2": 0, "imm": "0x6011ece1", "imm2": "0x89579f15", "rot": 11, "bit": 2, "mask": 4}, + {"i": 75, "op": "add", "dst": 1, "src": 3, "src2": 4, "imm": "0x3f4ffea2", "imm2": "0x8e35924e", "rot": 12, "bit": 15, "mask": 2}, + {"i": 76, "op": "mul", "dst": 1, "src": 3, "src2": 7, "imm": "0x15efa846", "imm2": "0xa1c971f1", "rot": 14, "bit": 20, "mask": 1}, + {"i": 77, "op": "mad", "dst": 0, "src": 3, "src2": 2, "imm": "0x8660530a", "imm2": "0xa013524b", "rot": 14, "bit": 12, "mask": 8}, + {"i": 78, "op": "shfl", "dst": 2, "src": 0, "src2": 0, "imm": "0xcc2e96bd", "imm2": "0xe32c6f87", "rot": 10, "bit": 11, "mask": 16}, + {"i": 79, "op": "mulhi", "dst": 1, "src": 0, "src2": 6, "imm": "0x61931e3f", "imm2": "0x86383e7f", "rot": 17, "bit": 3, "mask": 1}, + {"i": 80, "op": "mulhi", "dst": 0, "src": 4, "src2": 7, "imm": "0xd9991afc", "imm2": "0xd00e8678", "rot": 24, "bit": 15, "mask": 16}, + {"i": 81, "op": "mul", "dst": 4, "src": 2, "src2": 3, "imm": "0x472f70f4", "imm2": "0x9fde4236", "rot": 10, "bit": 5, "mask": 4}, + {"i": 82, "op": "sub", "dst": 5, "src": 7, "src2": 6, "imm": "0x55a7acb1", "imm2": "0x9fde8df1", "rot": 30, "bit": 29, "mask": 16}, + {"i": 83, "op": "mad", "dst": 0, "src": 2, "src2": 3, "imm": "0x1c487c40", "imm2": "0xd9d9a1c3", "rot": 9, "bit": 5, "mask": 4}, + {"i": 84, "op": "sub", "dst": 7, "src": 2, "src2": 4, "imm": "0x515c0a72", "imm2": "0x6df5c83d", "rot": 28, "bit": 21, "mask": 4}, + {"i": 85, "op": "xor", "dst": 1, "src": 2, "src2": 4, "imm": "0x917f1174", "imm2": "0x8e72e519", "rot": 9, "bit": 18, "mask": 16}, + {"i": 86, "op": "shfl", "dst": 6, "src": 3, "src2": 3, "imm": "0xda856cc2", "imm2": "0x932ae8d8", "rot": 28, "bit": 1, "mask": 16}, + {"i": 87, "op": "mad", "dst": 3, "src": 0, "src2": 0, "imm": "0x4ef22321", "imm2": "0x5666ada3", "rot": 4, "bit": 4, "mask": 1}, + {"i": 88, "op": "rotl", "dst": 0, "src": 1, "src2": 3, "imm": "0xbb3ec00c", "imm2": "0xffe78937", "rot": 20, "bit": 19, "mask": 4}, + {"i": 89, "op": "sub", "dst": 1, "src": 5, "src2": 4, "imm": "0x8f295708", "imm2": "0x6608577f", "rot": 17, "bit": 18, "mask": 16}, + {"i": 90, "op": "shfl", "dst": 6, "src": 4, "src2": 5, "imm": "0xd953aa15", "imm2": "0x4ef9f062", "rot": 19, "bit": 9, "mask": 16}, + {"i": 91, "op": "xor", "dst": 5, "src": 1, "src2": 5, "imm": "0x6c02cde2", "imm2": "0x118dbd2b", "rot": 2, "bit": 20, "mask": 16}, + {"i": 92, "op": "rotl", "dst": 5, "src": 0, "src2": 5, "imm": "0x9f0b0c61", "imm2": "0xe26dbe7a", "rot": 12, "bit": 17, "mask": 2}, + {"i": 93, "op": "mad", "dst": 6, "src": 1, "src2": 2, "imm": "0x78a44a91", "imm2": "0x3ce118a9", "rot": 18, "bit": 16, "mask": 16}, + {"i": 94, "op": "rotr", "dst": 2, "src": 5, "src2": 3, "imm": "0x02295fcb", "imm2": "0xc3353c64", "rot": 5, "bit": 31, "mask": 16}, + {"i": 95, "op": "or", "dst": 0, "src": 4, "src2": 4, "imm": "0x727120c3", "imm2": "0x528f4416", "rot": 6, "bit": 20, "mask": 8}, + {"i": 96, "op": "mul", "dst": 7, "src": 3, "src2": 4, "imm": "0x919c42ea", "imm2": "0x87526aa5", "rot": 28, "bit": 1, "mask": 8}, + {"i": 97, "op": "or", "dst": 5, "src": 0, "src2": 7, "imm": "0x8a0a100c", "imm2": "0xb2c65cc3", "rot": 13, "bit": 7, "mask": 4}, + {"i": 98, "op": "add", "dst": 3, "src": 1, "src2": 1, "imm": "0xbcfdd8a7", "imm2": "0x386bb642", "rot": 8, "bit": 13, "mask": 8}, + {"i": 99, "op": "add", "dst": 0, "src": 6, "src2": 7, "imm": "0xc921a021", "imm2": "0xa3962d03", "rot": 31, "bit": 9, "mask": 1}, + {"i": 100, "op": "rotr", "dst": 1, "src": 0, "src2": 3, "imm": "0x918c6f69", "imm2": "0x1a125801", "rot": 18, "bit": 29, "mask": 16}, + {"i": 101, "op": "add", "dst": 2, "src": 4, "src2": 1, "imm": "0x70da4067", "imm2": "0xc5990c5c", "rot": 3, "bit": 18, "mask": 4}, + {"i": 102, "op": "rotl", "dst": 2, "src": 7, "src2": 0, "imm": "0x9048d081", "imm2": "0xc25a9f56", "rot": 3, "bit": 25, "mask": 16}, + {"i": 103, "op": "shfl", "dst": 1, "src": 4, "src2": 7, "imm": "0x70ca1427", "imm2": "0x6632093d", "rot": 18, "bit": 28, "mask": 4}, + {"i": 104, "op": "xor", "dst": 0, "src": 2, "src2": 0, "imm": "0x0637204c", "imm2": "0x3b1b7f94", "rot": 1, "bit": 12, "mask": 2}, + {"i": 105, "op": "add", "dst": 4, "src": 3, "src2": 5, "imm": "0x1a5880cb", "imm2": "0xb15aec75", "rot": 31, "bit": 28, "mask": 1}, + {"i": 106, "op": "shfl", "dst": 4, "src": 0, "src2": 4, "imm": "0x7c22cb00", "imm2": "0xc4df7949", "rot": 3, "bit": 15, "mask": 2}, + {"i": 107, "op": "shfl", "dst": 7, "src": 0, "src2": 0, "imm": "0xd43c76ca", "imm2": "0x8d5b9751", "rot": 31, "bit": 28, "mask": 4}, + {"i": 108, "op": "mad", "dst": 1, "src": 7, "src2": 3, "imm": "0x69879993", "imm2": "0x3c4e18b0", "rot": 19, "bit": 25, "mask": 4}, + {"i": 109, "op": "mad", "dst": 0, "src": 2, "src2": 3, "imm": "0x8ce3a4a6", "imm2": "0x1b3f0b74", "rot": 19, "bit": 8, "mask": 2}, + {"i": 110, "op": "or", "dst": 2, "src": 4, "src2": 5, "imm": "0xcdbdac98", "imm2": "0x8d0e9f92", "rot": 19, "bit": 13, "mask": 1}, + {"i": 111, "op": "shfl", "dst": 6, "src": 1, "src2": 1, "imm": "0xe74157f6", "imm2": "0xf92dead6", "rot": 4, "bit": 14, "mask": 1}, + {"i": 112, "op": "mad", "dst": 2, "src": 5, "src2": 3, "imm": "0x200831bc", "imm2": "0x1ee98048", "rot": 29, "bit": 14, "mask": 1}, + {"i": 113, "op": "mulhi", "dst": 4, "src": 0, "src2": 7, "imm": "0xd6827ead", "imm2": "0xc0ee1933", "rot": 26, "bit": 23, "mask": 16}, + {"i": 114, "op": "rotl", "dst": 7, "src": 2, "src2": 2, "imm": "0x6ae1be2b", "imm2": "0x83216a4e", "rot": 29, "bit": 25, "mask": 16}, + {"i": 115, "op": "sub", "dst": 3, "src": 7, "src2": 1, "imm": "0xa1d1646e", "imm2": "0x6fa998ce", "rot": 15, "bit": 29, "mask": 2}, + {"i": 116, "op": "mulhi", "dst": 0, "src": 1, "src2": 4, "imm": "0xd7921c15", "imm2": "0x8b2345a7", "rot": 10, "bit": 19, "mask": 4}, + {"i": 117, "op": "rotl", "dst": 4, "src": 0, "src2": 4, "imm": "0xc1b1b0ab", "imm2": "0xd19b3f39", "rot": 12, "bit": 16, "mask": 1}, + {"i": 118, "op": "add", "dst": 1, "src": 4, "src2": 5, "imm": "0x24494ad3", "imm2": "0xc5e98ee5", "rot": 22, "bit": 7, "mask": 16}, + {"i": 119, "op": "add", "dst": 0, "src": 4, "src2": 6, "imm": "0x1b2d1c83", "imm2": "0xa97bc949", "rot": 8, "bit": 4, "mask": 16}, + {"i": 120, "op": "shfl", "dst": 5, "src": 4, "src2": 7, "imm": "0xb93015ff", "imm2": "0xc22366cd", "rot": 23, "bit": 16, "mask": 4}, + {"i": 121, "op": "shfl", "dst": 3, "src": 2, "src2": 5, "imm": "0xc0aaacdb", "imm2": "0xaaa7d229", "rot": 31, "bit": 29, "mask": 8}, + {"i": 122, "op": "rotl", "dst": 7, "src": 1, "src2": 3, "imm": "0xcce6a73f", "imm2": "0x517b6b86", "rot": 3, "bit": 16, "mask": 8}, + {"i": 123, "op": "rotr", "dst": 4, "src": 6, "src2": 0, "imm": "0x8e156aca", "imm2": "0x0a894b3e", "rot": 2, "bit": 12, "mask": 2}, + {"i": 124, "op": "shfl", "dst": 6, "src": 5, "src2": 2, "imm": "0xd88c8446", "imm2": "0xf49fedf3", "rot": 3, "bit": 23, "mask": 1}, + {"i": 125, "op": "rotl", "dst": 5, "src": 2, "src2": 0, "imm": "0xd10b27cf", "imm2": "0x2dde1faa", "rot": 21, "bit": 7, "mask": 2}, + {"i": 126, "op": "mad", "dst": 0, "src": 3, "src2": 1, "imm": "0x619e1818", "imm2": "0x62add310", "rot": 26, "bit": 1, "mask": 4}, + {"i": 127, "op": "add", "dst": 0, "src": 3, "src2": 6, "imm": "0xff14c08d", "imm2": "0xa7f6bb54", "rot": 25, "bit": 24, "mask": 4}, + {"i": 128, "op": "xor", "dst": 7, "src": 4, "src2": 4, "imm": "0x34224086", "imm2": "0x5e9f858e", "rot": 4, "bit": 16, "mask": 8}, + {"i": 129, "op": "mulhi", "dst": 3, "src": 5, "src2": 6, "imm": "0x89c3cbdd", "imm2": "0x4e80334c", "rot": 29, "bit": 4, "mask": 1}, + {"i": 130, "op": "shfl", "dst": 5, "src": 7, "src2": 1, "imm": "0x37d4c3e2", "imm2": "0xbd54378c", "rot": 14, "bit": 26, "mask": 8}, + {"i": 131, "op": "mulhi", "dst": 6, "src": 4, "src2": 3, "imm": "0x4f5fac7c", "imm2": "0x74070028", "rot": 5, "bit": 25, "mask": 8}, + {"i": 132, "op": "xor", "dst": 5, "src": 3, "src2": 0, "imm": "0x690eb003", "imm2": "0x8516a581", "rot": 20, "bit": 21, "mask": 2}, + {"i": 133, "op": "rotl", "dst": 0, "src": 5, "src2": 7, "imm": "0x42b31885", "imm2": "0x08249acb", "rot": 19, "bit": 6, "mask": 8}, + {"i": 134, "op": "rotr", "dst": 4, "src": 3, "src2": 3, "imm": "0x07d729ed", "imm2": "0xf564da8a", "rot": 17, "bit": 9, "mask": 16}, + {"i": 135, "op": "shfl", "dst": 1, "src": 2, "src2": 0, "imm": "0x19130ef2", "imm2": "0xf59815e5", "rot": 17, "bit": 12, "mask": 16}, + {"i": 136, "op": "sub", "dst": 5, "src": 2, "src2": 5, "imm": "0xdeb56168", "imm2": "0x871104c4", "rot": 13, "bit": 17, "mask": 16}, + {"i": 137, "op": "xor", "dst": 3, "src": 5, "src2": 4, "imm": "0x884fc7da", "imm2": "0x83a4a5bf", "rot": 23, "bit": 12, "mask": 4}, + {"i": 138, "op": "sub", "dst": 0, "src": 2, "src2": 6, "imm": "0x3e67a6fd", "imm2": "0x844d2039", "rot": 3, "bit": 24, "mask": 4}, + {"i": 139, "op": "mul", "dst": 3, "src": 7, "src2": 7, "imm": "0x965933b4", "imm2": "0xf37ef93d", "rot": 6, "bit": 9, "mask": 4}, + {"i": 140, "op": "add", "dst": 5, "src": 3, "src2": 3, "imm": "0x82ab98f4", "imm2": "0x2d465ca8", "rot": 11, "bit": 24, "mask": 1}, + {"i": 141, "op": "add", "dst": 4, "src": 1, "src2": 7, "imm": "0x80594390", "imm2": "0x237a9d8e", "rot": 14, "bit": 11, "mask": 8}, + {"i": 142, "op": "mad", "dst": 3, "src": 7, "src2": 0, "imm": "0xd37861a1", "imm2": "0x2fe70eae", "rot": 2, "bit": 11, "mask": 16}, + {"i": 143, "op": "or", "dst": 0, "src": 3, "src2": 5, "imm": "0x1a018e2c", "imm2": "0x301aa5f6", "rot": 15, "bit": 26, "mask": 1}, + {"i": 144, "op": "mulhi", "dst": 0, "src": 1, "src2": 5, "imm": "0x69d22dfc", "imm2": "0x4a90cb1e", "rot": 21, "bit": 9, "mask": 16}, + {"i": 145, "op": "xor", "dst": 6, "src": 2, "src2": 0, "imm": "0xf3f41f47", "imm2": "0xdf8d33e9", "rot": 22, "bit": 0, "mask": 8}, + {"i": 146, "op": "shfl", "dst": 7, "src": 1, "src2": 1, "imm": "0xf8a52377", "imm2": "0x0613ef41", "rot": 24, "bit": 12, "mask": 16}, + {"i": 147, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x07f15148", "imm2": "0xe68cf57e", "rot": 2, "bit": 18, "mask": 4}, + {"i": 148, "op": "or", "dst": 7, "src": 3, "src2": 0, "imm": "0xbfd39b49", "imm2": "0xc7f6e97c", "rot": 13, "bit": 27, "mask": 2}, + {"i": 149, "op": "xor", "dst": 4, "src": 2, "src2": 7, "imm": "0x47409e43", "imm2": "0x04ef7db9", "rot": 8, "bit": 6, "mask": 4}, + {"i": 150, "op": "xor", "dst": 7, "src": 1, "src2": 1, "imm": "0x5d42a156", "imm2": "0xc2f77c48", "rot": 27, "bit": 17, "mask": 4}, + {"i": 151, "op": "rotr", "dst": 5, "src": 2, "src2": 7, "imm": "0x7c170030", "imm2": "0xa3102916", "rot": 5, "bit": 6, "mask": 1}, + {"i": 152, "op": "mul", "dst": 0, "src": 7, "src2": 5, "imm": "0x16d009af", "imm2": "0x2a89533b", "rot": 26, "bit": 8, "mask": 4}, + {"i": 153, "op": "rotl", "dst": 4, "src": 5, "src2": 0, "imm": "0x6e54eb71", "imm2": "0xe2d27256", "rot": 16, "bit": 18, "mask": 1}, + {"i": 154, "op": "rotr", "dst": 6, "src": 4, "src2": 5, "imm": "0x128f24a4", "imm2": "0x59888463", "rot": 15, "bit": 17, "mask": 16}, + {"i": 155, "op": "xor", "dst": 6, "src": 1, "src2": 2, "imm": "0x34f276ff", "imm2": "0xef5086b4", "rot": 2, "bit": 10, "mask": 4}, + {"i": 156, "op": "or", "dst": 6, "src": 3, "src2": 4, "imm": "0x16cc2c2f", "imm2": "0xfae5a09b", "rot": 12, "bit": 31, "mask": 4}, + {"i": 157, "op": "add", "dst": 1, "src": 6, "src2": 7, "imm": "0x14878c5a", "imm2": "0x84c2a09d", "rot": 30, "bit": 19, "mask": 8}, + {"i": 158, "op": "sub", "dst": 4, "src": 1, "src2": 0, "imm": "0xfe18e4b0", "imm2": "0x239df0cc", "rot": 14, "bit": 20, "mask": 2}, + {"i": 159, "op": "mad", "dst": 4, "src": 6, "src2": 7, "imm": "0xbdfcc4e0", "imm2": "0x0870dee8", "rot": 1, "bit": 7, "mask": 8}, + {"i": 160, "op": "mul", "dst": 1, "src": 5, "src2": 1, "imm": "0xc837b43d", "imm2": "0x027ac950", "rot": 4, "bit": 22, "mask": 16}, + {"i": 161, "op": "or", "dst": 4, "src": 0, "src2": 2, "imm": "0x9c28f8f6", "imm2": "0x3db71fe6", "rot": 11, "bit": 6, "mask": 16}, + {"i": 162, "op": "rotl", "dst": 7, "src": 3, "src2": 6, "imm": "0xb6e56f55", "imm2": "0x5fcbd27c", "rot": 21, "bit": 10, "mask": 1}, + {"i": 163, "op": "sub", "dst": 0, "src": 1, "src2": 4, "imm": "0x8d95ae70", "imm2": "0xd4f10284", "rot": 15, "bit": 12, "mask": 8}, + {"i": 164, "op": "mul", "dst": 1, "src": 0, "src2": 6, "imm": "0xa9bdf99e", "imm2": "0x1f3dceb6", "rot": 21, "bit": 20, "mask": 16}, + {"i": 165, "op": "sub", "dst": 2, "src": 1, "src2": 4, "imm": "0xda245058", "imm2": "0x2038e7f6", "rot": 30, "bit": 11, "mask": 2}, + {"i": 166, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0xbbc8eee8", "imm2": "0xa73472fb", "rot": 23, "bit": 4, "mask": 2}, + {"i": 167, "op": "sub", "dst": 0, "src": 4, "src2": 2, "imm": "0x43540d69", "imm2": "0xb3f116ca", "rot": 28, "bit": 2, "mask": 2}, + {"i": 168, "op": "rotr", "dst": 1, "src": 6, "src2": 6, "imm": "0x57cea752", "imm2": "0xc1136e71", "rot": 25, "bit": 3, "mask": 8}, + {"i": 169, "op": "rotr", "dst": 7, "src": 1, "src2": 5, "imm": "0x4550a465", "imm2": "0x14362935", "rot": 25, "bit": 9, "mask": 16}, + {"i": 170, "op": "mad", "dst": 3, "src": 1, "src2": 3, "imm": "0x8c961d80", "imm2": "0x7de2e1db", "rot": 10, "bit": 6, "mask": 4}, + {"i": 171, "op": "shfl", "dst": 2, "src": 0, "src2": 0, "imm": "0xda047fca", "imm2": "0x459e17ab", "rot": 22, "bit": 26, "mask": 4}, + {"i": 172, "op": "sub", "dst": 2, "src": 1, "src2": 2, "imm": "0xbcc341fd", "imm2": "0x17a3b229", "rot": 8, "bit": 25, "mask": 8}, + {"i": 173, "op": "xor", "dst": 7, "src": 6, "src2": 7, "imm": "0x2302b424", "imm2": "0x78780982", "rot": 23, "bit": 7, "mask": 8}, + {"i": 174, "op": "rotr", "dst": 4, "src": 1, "src2": 2, "imm": "0xcad6fb5f", "imm2": "0x1807628e", "rot": 8, "bit": 21, "mask": 4}, + {"i": 175, "op": "shfl", "dst": 4, "src": 7, "src2": 7, "imm": "0xf5057e8e", "imm2": "0xd14d6cc5", "rot": 3, "bit": 12, "mask": 2}, + {"i": 176, "op": "mul", "dst": 6, "src": 2, "src2": 1, "imm": "0xc2926356", "imm2": "0xae74dc32", "rot": 1, "bit": 21, "mask": 8}, + {"i": 177, "op": "shfl", "dst": 0, "src": 2, "src2": 7, "imm": "0x8d94d98e", "imm2": "0x7bd6f4dc", "rot": 27, "bit": 14, "mask": 4}, + {"i": 178, "op": "add", "dst": 6, "src": 4, "src2": 6, "imm": "0x4fa43965", "imm2": "0x509871e6", "rot": 28, "bit": 21, "mask": 4}, + {"i": 179, "op": "xor", "dst": 0, "src": 6, "src2": 6, "imm": "0xe278fe6c", "imm2": "0x75d8a63f", "rot": 21, "bit": 15, "mask": 2}, + {"i": 180, "op": "add", "dst": 5, "src": 7, "src2": 7, "imm": "0xfb3c4bd8", "imm2": "0xb95b8a53", "rot": 22, "bit": 0, "mask": 4}, + {"i": 181, "op": "mul", "dst": 6, "src": 3, "src2": 2, "imm": "0x5a1347ec", "imm2": "0x3ccc5389", "rot": 27, "bit": 26, "mask": 4}, + {"i": 182, "op": "rotl", "dst": 7, "src": 4, "src2": 7, "imm": "0xdc6db2f7", "imm2": "0x7cdff0c6", "rot": 1, "bit": 2, "mask": 4}, + {"i": 183, "op": "add", "dst": 0, "src": 7, "src2": 7, "imm": "0xc448a197", "imm2": "0x7047b7cf", "rot": 24, "bit": 11, "mask": 8}, + {"i": 184, "op": "mul", "dst": 3, "src": 5, "src2": 0, "imm": "0xa5c0e492", "imm2": "0xaf7d8a86", "rot": 11, "bit": 16, "mask": 8}, + {"i": 185, "op": "shfl", "dst": 5, "src": 7, "src2": 1, "imm": "0x5965794b", "imm2": "0x36ddca6b", "rot": 7, "bit": 25, "mask": 1}, + {"i": 186, "op": "mul", "dst": 0, "src": 2, "src2": 5, "imm": "0xdfdbcc0f", "imm2": "0xac7b7091", "rot": 11, "bit": 31, "mask": 8}, + {"i": 187, "op": "mulhi", "dst": 4, "src": 7, "src2": 5, "imm": "0x3d8999e9", "imm2": "0xf1848db8", "rot": 15, "bit": 21, "mask": 8}, + {"i": 188, "op": "mad", "dst": 2, "src": 7, "src2": 2, "imm": "0xda3af48c", "imm2": "0x8e33a97c", "rot": 6, "bit": 30, "mask": 4}, + {"i": 189, "op": "sub", "dst": 7, "src": 2, "src2": 4, "imm": "0xa32da8c1", "imm2": "0xf7486d06", "rot": 25, "bit": 16, "mask": 4}, + {"i": 190, "op": "add", "dst": 2, "src": 5, "src2": 1, "imm": "0x2fceaf49", "imm2": "0x5caccc5d", "rot": 24, "bit": 19, "mask": 2}, + {"i": 191, "op": "shfl", "dst": 1, "src": 2, "src2": 2, "imm": "0x2a71f5e6", "imm2": "0xb9b51e89", "rot": 31, "bit": 25, "mask": 1}, + {"i": 192, "op": "add", "dst": 7, "src": 1, "src2": 7, "imm": "0xba0cee62", "imm2": "0x857bb5fe", "rot": 4, "bit": 20, "mask": 2}, + {"i": 193, "op": "rotl", "dst": 6, "src": 3, "src2": 3, "imm": "0xc7fe35cf", "imm2": "0x515c1201", "rot": 22, "bit": 7, "mask": 16}, + {"i": 194, "op": "xor", "dst": 3, "src": 7, "src2": 4, "imm": "0xe496d1f1", "imm2": "0x11046e3e", "rot": 16, "bit": 9, "mask": 4}, + {"i": 195, "op": "mul", "dst": 7, "src": 0, "src2": 2, "imm": "0xc7d5c4c2", "imm2": "0x077e940b", "rot": 23, "bit": 29, "mask": 4}, + {"i": 196, "op": "xor", "dst": 3, "src": 5, "src2": 3, "imm": "0xbe6db14a", "imm2": "0xf736da5a", "rot": 19, "bit": 6, "mask": 2}, + {"i": 197, "op": "add", "dst": 5, "src": 1, "src2": 5, "imm": "0x8519428c", "imm2": "0xeae84577", "rot": 18, "bit": 29, "mask": 1}, + {"i": 198, "op": "mad", "dst": 7, "src": 4, "src2": 4, "imm": "0xb272a029", "imm2": "0x071f6a5c", "rot": 5, "bit": 28, "mask": 8}, + {"i": 199, "op": "add", "dst": 3, "src": 6, "src2": 3, "imm": "0x468639d3", "imm2": "0x0bfdbfa1", "rot": 20, "bit": 8, "mask": 16}, + {"i": 200, "op": "mad", "dst": 5, "src": 3, "src2": 4, "imm": "0xffba18b7", "imm2": "0xbdcf9e81", "rot": 13, "bit": 31, "mask": 8}, + {"i": 201, "op": "shfl", "dst": 2, "src": 7, "src2": 7, "imm": "0xb7533191", "imm2": "0x8d58cc8c", "rot": 7, "bit": 29, "mask": 4}, + {"i": 202, "op": "add", "dst": 7, "src": 4, "src2": 6, "imm": "0x18ec9a69", "imm2": "0x53ee8e50", "rot": 12, "bit": 23, "mask": 16}, + {"i": 203, "op": "xor", "dst": 6, "src": 0, "src2": 5, "imm": "0x12011ff1", "imm2": "0xfbb57ac6", "rot": 16, "bit": 25, "mask": 16}, + {"i": 204, "op": "add", "dst": 4, "src": 5, "src2": 3, "imm": "0x3e81485b", "imm2": "0xac63376e", "rot": 1, "bit": 4, "mask": 1}, + {"i": 205, "op": "mad", "dst": 7, "src": 1, "src2": 4, "imm": "0xeb9eee5c", "imm2": "0x3f1daecd", "rot": 19, "bit": 29, "mask": 4}, + {"i": 206, "op": "shfl", "dst": 1, "src": 5, "src2": 2, "imm": "0x4fc50d06", "imm2": "0xb5533b31", "rot": 22, "bit": 27, "mask": 16}, + {"i": 207, "op": "mul", "dst": 4, "src": 0, "src2": 5, "imm": "0x202d6001", "imm2": "0x50388f85", "rot": 5, "bit": 18, "mask": 16}, + {"i": 208, "op": "or", "dst": 1, "src": 6, "src2": 7, "imm": "0x8743c0d4", "imm2": "0x9df69539", "rot": 14, "bit": 16, "mask": 1}, + {"i": 209, "op": "add", "dst": 6, "src": 4, "src2": 6, "imm": "0x66ccb75d", "imm2": "0x5b745519", "rot": 28, "bit": 28, "mask": 8}, + {"i": 210, "op": "rotr", "dst": 1, "src": 3, "src2": 1, "imm": "0xe9933132", "imm2": "0x6702fe85", "rot": 16, "bit": 19, "mask": 1}, + {"i": 211, "op": "sub", "dst": 5, "src": 4, "src2": 4, "imm": "0xfe04b942", "imm2": "0x40dd2ce8", "rot": 24, "bit": 8, "mask": 4}, + {"i": 212, "op": "xor", "dst": 4, "src": 7, "src2": 7, "imm": "0x10827878", "imm2": "0xef0de8fc", "rot": 17, "bit": 11, "mask": 8}, + {"i": 213, "op": "add", "dst": 1, "src": 7, "src2": 1, "imm": "0x0ce0553d", "imm2": "0x1c3dccf7", "rot": 26, "bit": 8, "mask": 1}, + {"i": 214, "op": "mulhi", "dst": 0, "src": 1, "src2": 1, "imm": "0xa71d7581", "imm2": "0xe567c74c", "rot": 2, "bit": 29, "mask": 2}, + {"i": 215, "op": "add", "dst": 3, "src": 6, "src2": 2, "imm": "0x96bfec89", "imm2": "0xfadc9205", "rot": 30, "bit": 9, "mask": 16}, + {"i": 216, "op": "rotl", "dst": 0, "src": 2, "src2": 5, "imm": "0x947ca474", "imm2": "0xd2a1b650", "rot": 4, "bit": 1, "mask": 1}, + {"i": 217, "op": "xor", "dst": 6, "src": 4, "src2": 5, "imm": "0x0b7617de", "imm2": "0xc6c0a9a2", "rot": 31, "bit": 15, "mask": 2}, + {"i": 218, "op": "shfl", "dst": 6, "src": 7, "src2": 1, "imm": "0x738f0a8f", "imm2": "0xfc868560", "rot": 2, "bit": 20, "mask": 2}, + {"i": 219, "op": "mad", "dst": 7, "src": 0, "src2": 1, "imm": "0x2960f2bc", "imm2": "0x7e9ef9b7", "rot": 21, "bit": 4, "mask": 16}, + {"i": 220, "op": "mul", "dst": 4, "src": 0, "src2": 2, "imm": "0x6014d800", "imm2": "0xd19ac7d1", "rot": 1, "bit": 13, "mask": 1}, + {"i": 221, "op": "mad", "dst": 2, "src": 6, "src2": 0, "imm": "0x4b558fc2", "imm2": "0x2d48bd3c", "rot": 18, "bit": 31, "mask": 2}, + {"i": 222, "op": "xor", "dst": 5, "src": 4, "src2": 6, "imm": "0x0260f3ab", "imm2": "0x8fce22f1", "rot": 21, "bit": 26, "mask": 16}, + {"i": 223, "op": "mul", "dst": 0, "src": 5, "src2": 0, "imm": "0xe07ab58f", "imm2": "0x5349a41d", "rot": 20, "bit": 3, "mask": 1}, + {"i": 224, "op": "add", "dst": 2, "src": 0, "src2": 5, "imm": "0x61495681", "imm2": "0x66b15c04", "rot": 24, "bit": 9, "mask": 16}, + {"i": 225, "op": "rotr", "dst": 3, "src": 2, "src2": 5, "imm": "0x291268bd", "imm2": "0xf046aa79", "rot": 27, "bit": 15, "mask": 4}, + {"i": 226, "op": "mad", "dst": 2, "src": 3, "src2": 3, "imm": "0xfd6dfa6c", "imm2": "0x2c0db93d", "rot": 29, "bit": 2, "mask": 16}, + {"i": 227, "op": "xor", "dst": 6, "src": 0, "src2": 4, "imm": "0xd4ae1ee2", "imm2": "0xe6a67972", "rot": 26, "bit": 1, "mask": 4}, + {"i": 228, "op": "rotl", "dst": 4, "src": 7, "src2": 3, "imm": "0xf6627a9d", "imm2": "0x99078da3", "rot": 30, "bit": 5, "mask": 16}, + {"i": 229, "op": "add", "dst": 2, "src": 3, "src2": 7, "imm": "0x70d05c34", "imm2": "0x25bc17c2", "rot": 20, "bit": 29, "mask": 4}, + {"i": 230, "op": "rotr", "dst": 1, "src": 5, "src2": 2, "imm": "0xef874bea", "imm2": "0x3bfbfecc", "rot": 17, "bit": 2, "mask": 2}, + {"i": 231, "op": "add", "dst": 1, "src": 4, "src2": 5, "imm": "0x32a28384", "imm2": "0x2d8728f3", "rot": 1, "bit": 8, "mask": 8}, + {"i": 232, "op": "rotl", "dst": 1, "src": 6, "src2": 3, "imm": "0x877c54ab", "imm2": "0x1d5e5014", "rot": 6, "bit": 24, "mask": 2}, + {"i": 233, "op": "add", "dst": 2, "src": 0, "src2": 5, "imm": "0xcf0949f1", "imm2": "0x61e81fe6", "rot": 2, "bit": 21, "mask": 16}, + {"i": 234, "op": "rotl", "dst": 4, "src": 5, "src2": 5, "imm": "0x5017c32c", "imm2": "0x1ab34dfe", "rot": 2, "bit": 1, "mask": 2}, + {"i": 235, "op": "sub", "dst": 2, "src": 6, "src2": 2, "imm": "0xadbdb811", "imm2": "0x2c2d4f2b", "rot": 30, "bit": 29, "mask": 8}, + {"i": 236, "op": "or", "dst": 6, "src": 3, "src2": 4, "imm": "0x700e204f", "imm2": "0x27294e7b", "rot": 13, "bit": 0, "mask": 4}, + {"i": 237, "op": "shfl", "dst": 7, "src": 5, "src2": 4, "imm": "0xc39f71ba", "imm2": "0x9a90263d", "rot": 28, "bit": 9, "mask": 4}, + {"i": 238, "op": "xor", "dst": 1, "src": 3, "src2": 6, "imm": "0x9f84a5b3", "imm2": "0x95691d15", "rot": 14, "bit": 22, "mask": 1}, + {"i": 239, "op": "rotl", "dst": 2, "src": 4, "src2": 0, "imm": "0xa3b79ed5", "imm2": "0x05f568bc", "rot": 1, "bit": 15, "mask": 4}, + {"i": 240, "op": "rotr", "dst": 5, "src": 2, "src2": 7, "imm": "0xaaabbb0d", "imm2": "0x71ce4cd6", "rot": 2, "bit": 26, "mask": 4}, + {"i": 241, "op": "rotl", "dst": 4, "src": 1, "src2": 7, "imm": "0x8146fbd3", "imm2": "0xde65fc31", "rot": 10, "bit": 9, "mask": 4}, + {"i": 242, "op": "add", "dst": 6, "src": 2, "src2": 7, "imm": "0xe204fb50", "imm2": "0x0a67565d", "rot": 5, "bit": 16, "mask": 1}, + {"i": 243, "op": "shfl", "dst": 1, "src": 3, "src2": 7, "imm": "0xcb7a3c1b", "imm2": "0x863f002c", "rot": 18, "bit": 1, "mask": 2}, + {"i": 244, "op": "shfl", "dst": 0, "src": 1, "src2": 3, "imm": "0x92eb030c", "imm2": "0x3cff4c6d", "rot": 1, "bit": 22, "mask": 16}, + {"i": 245, "op": "shfl", "dst": 5, "src": 7, "src2": 6, "imm": "0x046c969c", "imm2": "0xd21d7e76", "rot": 5, "bit": 31, "mask": 2}, + {"i": 246, "op": "xor", "dst": 4, "src": 5, "src2": 1, "imm": "0x2e3015f0", "imm2": "0x1b305a5c", "rot": 22, "bit": 16, "mask": 8}, + {"i": 247, "op": "xor", "dst": 7, "src": 2, "src2": 0, "imm": "0x8238cf71", "imm2": "0xb7874d94", "rot": 9, "bit": 1, "mask": 2}, + {"i": 248, "op": "shfl", "dst": 0, "src": 6, "src2": 7, "imm": "0x437cbc99", "imm2": "0x7fb41fed", "rot": 17, "bit": 1, "mask": 16}, + {"i": 249, "op": "mulhi", "dst": 1, "src": 3, "src2": 2, "imm": "0xeb293d88", "imm2": "0xb92f5968", "rot": 3, "bit": 20, "mask": 2}, + {"i": 250, "op": "add", "dst": 7, "src": 5, "src2": 3, "imm": "0x3d6daffd", "imm2": "0x5a79fd6f", "rot": 6, "bit": 31, "mask": 2}, + {"i": 251, "op": "add", "dst": 6, "src": 0, "src2": 3, "imm": "0x84334aea", "imm2": "0x6beb28c2", "rot": 27, "bit": 28, "mask": 2}, + {"i": 252, "op": "sub", "dst": 7, "src": 1, "src2": 6, "imm": "0xe6177638", "imm2": "0x814d3fc5", "rot": 25, "bit": 2, "mask": 2}, + {"i": 253, "op": "rotr", "dst": 5, "src": 2, "src2": 7, "imm": "0x95560bac", "imm2": "0xbf8b01a9", "rot": 14, "bit": 5, "mask": 16}, + {"i": 254, "op": "mulhi", "dst": 2, "src": 5, "src2": 1, "imm": "0x318956e5", "imm2": "0x7c20b373", "rot": 13, "bit": 30, "mask": 2}, + {"i": 255, "op": "mul", "dst": 6, "src": 3, "src2": 3, "imm": "0xe1998e49", "imm2": "0x59353e5a", "rot": 15, "bit": 2, "mask": 4} + ]}, + "instructions": [ + {"i": 0, "op": "sub", "dst": 6, "src": 3, "src2": 3, "imm": "0x77dfbe60", "imm2": "0x404c8b9c", "rot": 24, "bit": 8, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 1, "op": "or", "dst": 2, "src": 6, "src2": 7, "imm": "0xb2e79058", "imm2": "0xc2485073", "rot": 21, "bit": 23, "mask": 8, "width": 1, "win": 0, "off": 0}, + {"i": 2, "op": "add", "dst": 4, "src": 0, "src2": 6, "imm": "0x1a579b38", "imm2": "0x7731324b", "rot": 6, "bit": 4, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 3, "op": "load", "dst": 3, "src": 6, "src2": 0, "imm": "0x683f5bc5", "imm2": "0xc8a218fe", "rot": 25, "bit": 4, "mask": 1, "width": 1, "win": 0, "off": 0}, + {"i": 4, "op": "shfl", "dst": 0, "src": 6, "src2": 4, "imm": "0x4d7cdcbb", "imm2": "0x1a5bf1a4", "rot": 8, "bit": 28, "mask": 1, "width": 1, "win": 0, "off": 0}, + {"i": 5, "op": "xor", "dst": 4, "src": 1, "src2": 0, "imm": "0xa2b86827", "imm2": "0x94a44439", "rot": 21, "bit": 22, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 6, "op": "mad", "dst": 3, "src": 5, "src2": 3, "imm": "0x58a4ea3f", "imm2": "0x467879fa", "rot": 6, "bit": 31, "mask": 8, "width": 1, "win": 0, "off": 0}, + {"i": 7, "op": "mul", "dst": 1, "src": 6, "src2": 1, "imm": "0x31fcc19c", "imm2": "0x176cb88f", "rot": 1, "bit": 11, "mask": 1, "width": 1, "win": 0, "off": 0}, + {"i": 8, "op": "load", "dst": 2, "src": 0, "src2": 3, "imm": "0x89b12386", "imm2": "0x661ed5e5", "rot": 11, "bit": 13, "mask": 2, "width": 1, "win": 2, "off": 2}, + {"i": 9, "op": "xor", "dst": 0, "src": 4, "src2": 1, "imm": "0x69150105", "imm2": "0x8e1f04b9", "rot": 3, "bit": 30, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 10, "op": "add", "dst": 0, "src": 7, "src2": 5, "imm": "0xee083919", "imm2": "0xb10fcef8", "rot": 9, "bit": 30, "mask": 16, "width": 1, "win": 0, "off": 0}, + {"i": 11, "op": "mulhi", "dst": 0, "src": 2, "src2": 3, "imm": "0x99ac7c83", "imm2": "0x34eb2845", "rot": 18, "bit": 12, "mask": 16, "width": 1, "win": 0, "off": 0}, + {"i": 12, "op": "xor", "dst": 5, "src": 4, "src2": 0, "imm": "0x97cfc887", "imm2": "0x0458f053", "rot": 28, "bit": 24, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 13, "op": "mad", "dst": 7, "src": 3, "src2": 0, "imm": "0x503e87b4", "imm2": "0x2c5d2617", "rot": 3, "bit": 15, "mask": 1, "width": 1, "win": 0, "off": 0}, + {"i": 14, "op": "load", "dst": 3, "src": 4, "src2": 4, "imm": "0xfae98035", "imm2": "0x6c127281", "rot": 19, "bit": 29, "mask": 4, "width": 1, "win": 1, "off": 0}, + {"i": 15, "op": "load", "dst": 2, "src": 5, "src2": 2, "imm": "0x67601ed5", "imm2": "0x3c546ed1", "rot": 28, "bit": 28, "mask": 4, "width": 1, "win": 0, "off": 0}, + {"i": 16, "op": "shfl", "dst": 7, "src": 5, "src2": 6, "imm": "0x7789be79", "imm2": "0x87496b8e", "rot": 22, "bit": 2, "mask": 16, "width": 1, "win": 0, "off": 0}, + {"i": 17, "op": "mul", "dst": 7, "src": 6, "src2": 2, "imm": "0x559099bb", "imm2": "0x6a72e1a6", "rot": 1, "bit": 11, "mask": 16, "width": 1, "win": 0, "off": 0}, + {"i": 18, "op": "mad", "dst": 2, "src": 3, "src2": 1, "imm": "0x4f5bae5e", "imm2": "0x4c8c5762", "rot": 27, "bit": 31, "mask": 4, "width": 1, "win": 0, "off": 0}, + {"i": 19, "op": "rotr", "dst": 5, "src": 2, "src2": 4, "imm": "0x5dc5e6bf", "imm2": "0x36bc207d", "rot": 9, "bit": 4, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 20, "op": "load", "dst": 5, "src": 3, "src2": 7, "imm": "0x4a278885", "imm2": "0xcd1044b0", "rot": 30, "bit": 0, "mask": 4, "width": 1, "win": 1, "off": 1}, + {"i": 21, "op": "shfl", "dst": 6, "src": 4, "src2": 3, "imm": "0x48b94ffc", "imm2": "0xd8e14d80", "rot": 2, "bit": 31, "mask": 16, "width": 1, "win": 0, "off": 0}, + {"i": 22, "op": "mulhi", "dst": 1, "src": 2, "src2": 7, "imm": "0x47bd8492", "imm2": "0x28fa8d0b", "rot": 19, "bit": 26, "mask": 4, "width": 1, "win": 0, "off": 0}, + {"i": 23, "op": "xor", "dst": 6, "src": 5, "src2": 1, "imm": "0x4ed6baf4", "imm2": "0xa8c31f74", "rot": 30, "bit": 2, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 24, "op": "mad", "dst": 5, "src": 2, "src2": 0, "imm": "0x6174747d", "imm2": "0xfc912524", "rot": 18, "bit": 29, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 25, "op": "mad", "dst": 6, "src": 3, "src2": 7, "imm": "0x113a6442", "imm2": "0xbe52ec57", "rot": 15, "bit": 15, "mask": 1, "width": 1, "win": 0, "off": 0}, + {"i": 26, "op": "load", "dst": 6, "src": 5, "src2": 1, "imm": "0x9d3eb1cc", "imm2": "0x375c73bb", "rot": 29, "bit": 7, "mask": 8, "width": 1, "win": 1, "off": 1}, + {"i": 27, "op": "mad", "dst": 1, "src": 5, "src2": 5, "imm": "0x150088c3", "imm2": "0x7f17e789", "rot": 8, "bit": 25, "mask": 8, "width": 1, "win": 0, "off": 0}, + {"i": 28, "op": "load", "dst": 0, "src": 6, "src2": 3, "imm": "0x4b412223", "imm2": "0x9192ceeb", "rot": 16, "bit": 28, "mask": 1, "width": 1, "win": 1, "off": 1}, + {"i": 29, "op": "add", "dst": 7, "src": 4, "src2": 6, "imm": "0xd5e6c37c", "imm2": "0x0b1f02c7", "rot": 2, "bit": 7, "mask": 8, "width": 1, "win": 0, "off": 0}, + {"i": 30, "op": "add", "dst": 7, "src": 0, "src2": 6, "imm": "0x4e45ba51", "imm2": "0x2d6e8bd1", "rot": 7, "bit": 0, "mask": 8, "width": 1, "win": 0, "off": 0}, + {"i": 31, "op": "xor", "dst": 0, "src": 2, "src2": 7, "imm": "0xef601d89", "imm2": "0x805402db", "rot": 7, "bit": 28, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 32, "op": "add", "dst": 6, "src": 2, "src2": 3, "imm": "0x1c91b2a1", "imm2": "0x351422d2", "rot": 1, "bit": 13, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 33, "op": "mulhi", "dst": 4, "src": 5, "src2": 7, "imm": "0x7cf21567", "imm2": "0x1075ee0d", "rot": 26, "bit": 17, "mask": 16, "width": 1, "win": 0, "off": 0}, + {"i": 34, "op": "rotr", "dst": 3, "src": 5, "src2": 6, "imm": "0x60d8f191", "imm2": "0x40523707", "rot": 3, "bit": 10, "mask": 4, "width": 1, "win": 0, "off": 0}, + {"i": 35, "op": "load", "dst": 3, "src": 0, "src2": 5, "imm": "0xb1dc3f6e", "imm2": "0x74440531", "rot": 21, "bit": 10, "mask": 1, "width": 1, "win": 0, "off": 0}, + {"i": 36, "op": "rotl", "dst": 4, "src": 6, "src2": 4, "imm": "0xf0d17569", "imm2": "0xd281fd01", "rot": 13, "bit": 4, "mask": 8, "width": 1, "win": 0, "off": 0}, + {"i": 37, "op": "add", "dst": 6, "src": 1, "src2": 6, "imm": "0x3f2970f7", "imm2": "0x64a28251", "rot": 4, "bit": 7, "mask": 8, "width": 1, "win": 0, "off": 0}, + {"i": 38, "op": "mul", "dst": 2, "src": 0, "src2": 3, "imm": "0x6463770c", "imm2": "0xb85d603f", "rot": 27, "bit": 1, "mask": 1, "width": 1, "win": 0, "off": 0}, + {"i": 39, "op": "add", "dst": 3, "src": 4, "src2": 3, "imm": "0x3b7b3317", "imm2": "0x7378c955", "rot": 28, "bit": 15, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 40, "op": "load", "dst": 0, "src": 7, "src2": 3, "imm": "0xd0fb5982", "imm2": "0x2e4d31a8", "rot": 25, "bit": 27, "mask": 4, "width": 1, "win": 2, "off": 3}, + {"i": 41, "op": "xor", "dst": 6, "src": 5, "src2": 0, "imm": "0xe96a9e6e", "imm2": "0x8d73cc3e", "rot": 17, "bit": 11, "mask": 16, "width": 1, "win": 0, "off": 0}, + {"i": 42, "op": "add", "dst": 3, "src": 0, "src2": 6, "imm": "0x6678b059", "imm2": "0xb31f6a64", "rot": 11, "bit": 2, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 43, "op": "load", "dst": 7, "src": 3, "src2": 2, "imm": "0x65a841c5", "imm2": "0x6aae1403", "rot": 9, "bit": 9, "mask": 1, "width": 1, "win": 1, "off": 0}, + {"i": 44, "op": "shfl", "dst": 3, "src": 7, "src2": 6, "imm": "0x650ba38d", "imm2": "0x07218508", "rot": 12, "bit": 13, "mask": 4, "width": 1, "win": 0, "off": 0}, + {"i": 45, "op": "mulhi", "dst": 1, "src": 5, "src2": 7, "imm": "0x18736788", "imm2": "0x9f820eed", "rot": 13, "bit": 29, "mask": 1, "width": 1, "win": 0, "off": 0}, + {"i": 46, "op": "rotl", "dst": 0, "src": 1, "src2": 5, "imm": "0x48deba75", "imm2": "0xefc508fa", "rot": 6, "bit": 6, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 47, "op": "load", "dst": 1, "src": 7, "src2": 4, "imm": "0x28a04384", "imm2": "0xc67cb0ac", "rot": 5, "bit": 5, "mask": 2, "width": 1, "win": 1, "off": 1}, + {"i": 48, "op": "add", "dst": 4, "src": 0, "src2": 5, "imm": "0x43095946", "imm2": "0x5feccee6", "rot": 22, "bit": 17, "mask": 1, "width": 1, "win": 0, "off": 0}, + {"i": 49, "op": "load", "dst": 2, "src": 0, "src2": 2, "imm": "0x3e78cdd8", "imm2": "0x7c170311", "rot": 10, "bit": 5, "mask": 16, "width": 1, "win": 0, "off": 0}, + {"i": 50, "op": "sub", "dst": 2, "src": 3, "src2": 7, "imm": "0xaff0cdb2", "imm2": "0xde0b7ee0", "rot": 21, "bit": 11, "mask": 8, "width": 1, "win": 0, "off": 0}, + {"i": 51, "op": "add", "dst": 7, "src": 2, "src2": 1, "imm": "0xf45ecdf8", "imm2": "0x26e2b582", "rot": 28, "bit": 24, "mask": 4, "width": 1, "win": 0, "off": 0}, + {"i": 52, "op": "load", "dst": 5, "src": 7, "src2": 7, "imm": "0x67353e69", "imm2": "0xb2d56082", "rot": 27, "bit": 14, "mask": 2, "width": 1, "win": 2, "off": 3}, + {"i": 53, "op": "load", "dst": 6, "src": 3, "src2": 1, "imm": "0x5a58c787", "imm2": "0x53a0467b", "rot": 6, "bit": 16, "mask": 16, "width": 1, "win": 0, "off": 0}, + {"i": 54, "op": "shfl", "dst": 7, "src": 5, "src2": 4, "imm": "0x3c237ad5", "imm2": "0xf2681187", "rot": 14, "bit": 26, "mask": 4, "width": 1, "win": 0, "off": 0}, + {"i": 55, "op": "rotr", "dst": 1, "src": 2, "src2": 3, "imm": "0x1c25c077", "imm2": "0x125de1ff", "rot": 13, "bit": 13, "mask": 2, "width": 1, "win": 0, "off": 0}, + {"i": 56, "op": "shfl", "dst": 0, "src": 5, "src2": 0, "imm": "0x6465afe2", "imm2": "0x23b41d56", "rot": 31, "bit": 28, "mask": 16, "width": 1, "win": 0, "off": 0}, + {"i": 57, "op": "rotl", "dst": 5, "src": 0, "src2": 2, "imm": "0xa427ff6e", "imm2": "0x014a0deb", "rot": 12, "bit": 28, "mask": 8, "width": 1, "win": 0, "off": 0}, + {"i": 58, "op": "add", "dst": 5, "src": 3, "src2": 7, "imm": "0x6b4f35e8", "imm2": "0x473b1718", "rot": 25, "bit": 4, "mask": 8, "width": 1, "win": 0, "off": 0}, + {"i": 59, "op": "rotr", "dst": 4, "src": 1, "src2": 3, "imm": "0x66cc96ef", "imm2": "0xcb8c8e56", "rot": 7, "bit": 6, "mask": 4, "width": 1, "win": 0, "off": 0}, + {"i": 60, "op": "shfl", "dst": 4, "src": 5, "src2": 5, "imm": "0x461bbf9f", "imm2": "0xc1ffd350", "rot": 19, "bit": 2, "mask": 1, "width": 1, "win": 0, "off": 0}, + {"i": 61, "op": "load", "dst": 7, "src": 0, "src2": 1, "imm": "0x4d893571", "imm2": "0x5583e644", "rot": 19, "bit": 19, "mask": 8, "width": 1, "win": 1, "off": 1}, + {"i": 62, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x7a9049b4", "imm2": "0xbb7ebdf8", "rot": 15, "bit": 11, "mask": 8, "width": 1, "win": 1, "off": 1}, + {"i": 63, "op": "mul", "dst": 4, "src": 1, "src2": 7, "imm": "0xe7ae8a83", "imm2": "0x1ded64e8", "rot": 21, "bit": 1, "mask": 8, "width": 1, "win": 0, "off": 0} + ] +} diff --git a/tools/attack/adv-accept-v5/inputs/v5-dn3-epoch0-state.igsd1 b/tools/attack/adv-accept-v5/inputs/v5-dn3-epoch0-state.igsd1 new file mode 100644 index 000000000..7bd85356f Binary files /dev/null and b/tools/attack/adv-accept-v5/inputs/v5-dn3-epoch0-state.igsd1 differ diff --git a/tools/attack/adv-accept-v5/src/live.rs b/tools/attack/adv-accept-v5/src/live.rs new file mode 100644 index 000000000..689d3a6f0 --- /dev/null +++ b/tools/attack/adv-accept-v5/src/live.rs @@ -0,0 +1,1770 @@ +//! attack-f8: the uniformity censuses of attack-pass row F8 (`docs/plans/cryptanalysis.md` 4.2 F8) on the real +//! class v4 derivation through the `igneum-pow` library. +//! +//! Three censuses, one binary: +//! +//! * `lines`: the cache-line index `s[0] AND line_mask` (MEMHARD.md 1.6, 2^22 lines) of every one of the 8 reads of +//! every item `t < 2^items_log2` on `days` consecutive day keys: the full 2^22-line histogram per round and +//! pooled, the 2^16-bucket (64-line segment) histogram, the largest bucket in sigma, chi-square, and a uniform +//! SplitMix64 control of the same size. Every item is also derived by the library's `derive_items` and compared +//! word for word (the traced mirror is trusted only while that holds). +//! * `warps`: the warp interpreter mirrored with every load's item index and the item's 8 lines recorded (the +//! day's items derived once into a table, as a GPU holds the dataset): distinct lines and items per hash and per +//! warp over `nonces` nonces of one program, the cross-hash item histogram with the hot-set test, and a uniform +//! control. Sampled warps are hashed by `Epoch::hash_warp` as well and must agree bit for bit. +//! +//! Plant hooks (the known-fail firings): `--plant quarter-lines` masks the line index to a quarter of its range, +//! `--plant half-lines` to a half, `--plant const-item` feeds one constant item at the program's first load site. +//! Under a plant the library comparisons are skipped (the plant is not the derivation) and the log says so. +//! +//! Nothing in `igneum-pow` is modified; every constant and the read sequence are the library's. + +use igneum_pow::bind::{day_bytes, hex, unhex}; +use igneum_pow::generator::{EraParams, Instr, Op, Program, ProgramClass, ITERATIONS, LANES}; +use igneum_pow::memhard::{mixer, round_key_mult, Cache, Layout, MixParams, ITEM_ROUNDS}; +use igneum_pow::seed::{seed_words_from_bytes, SplitMix64}; +use igneum_pow::verify::{load_index, splitmix32, DatasetSource, Epoch}; +use igneum_pow::state::{StateLeaves, StateStream}; +use std::sync::{Arc, OnceLock}; + +/// adv-live-v5 (adv-accept lane, 7 October 2026): the f8 harness over CLASS V5 (vendored igneum-pow of branch class-v5): +/// the same program draw (the state flag is not read by the draw), the dataset keyed by the window's state leaves read +/// from `--state ` (the v5 Devnet 3 pack's state.igsd1). Everything else is the f8 harness unchanged. +static STATE: OnceLock> = OnceLock::new(); +fn state_leaves() -> Arc { + STATE.get().expect("--state is required under class v5").clone() +} +use std::io::Write; +use std::sync::atomic::{AtomicU32, AtomicUsize, Ordering}; +use std::time::{Instant, SystemTime, UNIX_EPOCH}; + +/// The devnet genesis hash: epoch seed and era seed of the chain's epoch 0 (`igneum-pow/tests/packs.rs`). +const GENESIS_HEX: &str = "edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07"; +/// The devnet pack's day index (`bind::day_bytes(20730)`, 2026-10-04). +const DEFAULT_DAY: u64 = 20730; +const DATASET_LOG2: u32 = 28; + +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +enum Plant { + None, + QuarterLines, + HalfLines, + ConstItem, +} + +impl Plant { + fn parse(s: &str) -> Plant { + match s { + "none" => Plant::None, + "quarter-lines" => Plant::QuarterLines, + "half-lines" => Plant::HalfLines, + "const-item" => Plant::ConstItem, + _ => panic!("unknown plant {s}"), + } + } + fn line_mask(self) -> u32 { + match self { + Plant::QuarterLines => (1u32 << 20) - 1, + Plant::HalfLines => (1u32 << 21) - 1, + _ => u32::MAX, + } + } + fn name(self) -> &'static str { + match self { + Plant::None => "none", + Plant::QuarterLines => "quarter-lines", + Plant::HalfLines => "half-lines", + Plant::ConstItem => "const-item", + } + } +} + +fn utc_now() -> String { + let s = SystemTime::now().duration_since(UNIX_EPOCH).unwrap().as_secs(); + let (d, t) = (s / 86400, s % 86400); + // civil date from days (Howard Hinnant's algorithm) + let z = d as i64 + 719468; + let era = z.div_euclid(146097); + let doe = z.rem_euclid(146097); + let yoe = (doe - doe / 1460 + doe / 36524 - doe / 146096) / 365; + let y = yoe + era * 400; + let doy = doe - (365 * yoe + yoe / 4 - yoe / 100); + let mp = (5 * doy + 2) / 153; + let dd = doy - (153 * mp + 2) / 5 + 1; + let mm = if mp < 10 { mp + 3 } else { mp - 9 }; + let yy = if mm <= 2 { y + 1 } else { y }; + format!("{yy:04}-{mm:02}-{dd:02}T{:02}:{:02}:{:02}Z", t / 3600, (t / 60) % 60, t % 60) +} + +macro_rules! log { + ($($arg:tt)*) => {{ + println!("[{}] {}", utc_now(), format!($($arg)*)); + std::io::stdout().flush().ok(); + }}; +} + +// -------------------------------------------------------------------------------------------------------------- +// The traced item derivation: `memhard::derive_items_mask` instruction for instruction, with the line index of +// every round recorded. Batched like the library so the 8 dependent misses of independent items overlap. +// -------------------------------------------------------------------------------------------------------------- + +fn derive_traced(ts: &[u32], mp: &MixParams, cache: &Cache, out: &mut [[u32; 16]], lines: &mut [[u32; 8]], plant_mask: u32) { + let leaves = state_leaves(); + let n = ts.len(); + let m = mp.shape.mixer_mult as usize; + assert_eq!(mp.shape.derive_len, 0, "class v4 has the fixed mixer"); + let line_mask = cache.line_mask(); + for k in 0..n { + let s = &mut out[k]; + let t = ts[k]; + s[..8].copy_from_slice(&mp.key); + for i in 0..8 { + s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]); + } + // class v5: the window's state leaf of item t, XORed into the 16 initial words before the first mixer + let leaf = leaves.leaf(t); + for i in 0..16 { + s[i] ^= leaf[i]; + } + } + for r in 0..ITEM_ROUNDS { + for j in 0..m { + let rk = round_key_mult(r, j, m); + for s in out[..n].iter_mut() { + mixer(s, rk, mp); + } + } + for k in 0..n { + let s = &mut out[k]; + // MEMHARD.md 1.6: `a = s[0] AND 0x003fffff`, the cache line index (the mask is the cache's at every size) + let a = s[0] & line_mask & plant_mask; + lines[k][r] = a; + let line = cache.line(a); + for i in 0..16 { + s[i] ^= line[i]; + } + } + } + for j in 0..m { + let rk = round_key_mult(ITEM_ROUNDS, j, m); + for s in out[..n].iter_mut() { + mixer(s, rk, mp); + } + } +} + +// -------------------------------------------------------------------------------------------------------------- +// Histogram statistics +// -------------------------------------------------------------------------------------------------------------- + +struct Stats { + bins: usize, + total: u64, + mean: f64, + sigma: f64, + max: u64, + argmax: usize, + min: u64, + z_max: f64, + z_min: f64, + chi2_per_dof: f64, + chi2_z: f64, + /// (fraction f, top-f share S_f) for f = 1/1000, 1/200, 1/100 + top: [(f64, f64); 3], +} + +fn stats_of(counts: impl Iterator + Clone, bins: usize) -> Stats { + let total: u64 = counts.clone().sum(); + let mean = total as f64 / bins as f64; + let sigma = mean.sqrt(); + let mut max = 0u64; + let mut argmax = 0usize; + let mut min = u64::MAX; + let mut chi2 = 0f64; + for (i, c) in counts.clone().enumerate() { + if c > max { + max = c; + argmax = i; + } + if c < min { + min = c; + } + let d = c as f64 - mean; + chi2 += d * d; + } + chi2 /= mean; + let dof = (bins - 1) as f64; + // count of counts for the top-f shares (counts at or above the cap, rare, are sorted on their own) + let cap = 1usize << 16; + let mut coc = vec![0u64; cap]; + let mut big: Vec = Vec::new(); + for c in counts { + if (c as usize) < cap { + coc[c as usize] += 1; + } else { + big.push(c); + } + } + big.sort_unstable_by(|a, b| b.cmp(a)); + let mut top = [(0f64, 0f64); 3]; + for (k, f) in [1.0 / 1000.0, 1.0 / 200.0, 1.0 / 100.0].into_iter().enumerate() { + let want = (f * bins as f64).round() as u64; + let mut left = want; + let mut reads = 0u64; + for &c in &big { + if left == 0 { + break; + } + reads += c; + left -= 1; + } + let mut v = cap - 1; + while left > 0 { + let n = coc[v].min(left); + reads += n * v as u64; + left -= n; + if v == 0 { + break; + } + v -= 1; + } + top[k] = (f, reads as f64 / total.max(1) as f64); + } + Stats { + bins, + total, + mean, + sigma, + max, + argmax, + min, + z_max: (max as f64 - mean) / sigma, + z_min: (min as f64 - mean) / sigma, + chi2_per_dof: chi2 / dof, + chi2_z: (chi2 - dof) / (2.0 * dof).sqrt(), + top, + } +} + +impl Stats { + fn line(&self, label: &str) -> String { + format!( + "{label}: bins {} reads {} mean {:.3} sigma {:.3} max {} (bin {}) z_max {:+.2} min {} z_min {:+.2} chi2/dof {:.5} chi2_z {:+.2} top0.1% {:.5}% top0.5% {:.5}% top1% {:.5}%", + self.bins, + self.total, + self.mean, + self.sigma, + self.max, + self.argmax, + self.z_max, + self.min, + self.z_min, + self.chi2_per_dof, + self.chi2_z, + self.top[0].1 * 100.0, + self.top[1].1 * 100.0, + self.top[2].1 * 100.0 + ) + } +} + +fn snapshot(counts: &[AtomicU32]) -> Vec { + counts.iter().map(|x| x.load(Ordering::Relaxed) as u64).collect() +} + +fn zero(counts: &[AtomicU32]) { + for c in counts { + c.store(0, Ordering::Relaxed); + } +} + +fn atomic_vec(n: usize) -> Vec { + (0..n).map(|_| AtomicU32::new(0)).collect() +} + +/// A uniform control: `n` SplitMix64 indices into `bins` bins, `threads` threads. +fn control(bins: usize, n: u64, seed: u64, threads: usize, hist: &[AtomicU32]) { + let mask = (bins - 1) as u64; + assert!(bins.is_power_of_two()); + std::thread::scope(|sc| { + for th in 0..threads { + let hist = &hist; + sc.spawn(move || { + let mut s = SplitMix64::new(seed ^ (th as u64).wrapping_mul(0x9E3779B97F4A7C15)); + let per = n / threads as u64 + if (th as u64) < n % threads as u64 { 1 } else { 0 }; + for _ in 0..per { + let i = (s.next() & mask) as usize; + hist[i].fetch_add(1, Ordering::Relaxed); + } + }); + } + }); +} + +/// The hot-set test (F8's definition, reused by F9): for the top-f items by count, the share S_f of all reads +/// they receive, against the same share E_f of a uniform control of the same size; the excess X_f = S_f - E_f. +/// The set is a hot set when X_f >= f for any f in {0.1%, 0.5%, 1%}: after the chance excess is removed, the top +/// f of items capture at least one extra proportional share (at least 2f of the reads above chance, which is +/// what a chip's on-die copy of f of the items would have to win to matter). The statistical sensitivity is +/// printed beside it: X_f in units of f. +fn hot_set_test(label: &str, real: &Stats, ctrl: &Stats) -> bool { + let mut flagged = false; + for k in 0..3 { + let (f, s) = real.top[k]; + let e = ctrl.top[k].1; + let x = s - e; + let fire = x >= f; + flagged |= fire; + log!( + "hot-set {label}: f {:.1}% S_f {:.5}% E_f(control) {:.5}% X_f {:+.5}% X_f/f {:+.4} -> {}", + f * 100.0, + s * 100.0, + e * 100.0, + x * 100.0, + x / f, + if fire { "HOT SET" } else { "no hot set" } + ); + } + log!("hot-set {label}: verdict {}", if flagged { "FLAGGED" } else { "clear" }); + flagged +} + +/// The 6-sigma test: the largest bucket within 6 sigma of uniform. +fn six_sigma_test(label: &str, s: &Stats) -> bool { + let fire = s.z_max > 6.0 || s.z_min < -6.0; + log!( + "6-sigma {label}: largest bucket {} at {:+.2} sigma, smallest {} at {:+.2} sigma -> {}", + s.max, + s.z_max, + s.min, + s.z_min, + if fire { "FLAGGED (beyond 6 sigma)" } else { "within 6 sigma" } + ); + fire +} + +// -------------------------------------------------------------------------------------------------------------- +// Census 1: the line index over all items of `days` day keys +// -------------------------------------------------------------------------------------------------------------- + +fn census_lines(day0: u64, days: u64, items_log2: u32, threads: usize, plant: Plant, validate: &str, out_dir: &str) { + log!("census lines: day0 {day0} days {days} items 2^{items_log2} per day threads {threads} plant {} validate {validate}", plant.name()); + let lines_n = 1usize << 22; + let per_round: Vec> = (0..ITEM_ROUNDS).map(|_| atomic_vec(lines_n)).collect(); + let pooled: Vec> = (0..ITEM_ROUNDS).map(|_| atomic_vec(lines_n)).collect(); + let items = 1u64 << items_log2; + let batch = 64u64; + let mut any_fire = false; + let mut mismatches_total = 0u64; + let mut derivations = 0u64; + let t_all = Instant::now(); + for d in day0..day0 + days { + let t0 = Instant::now(); + let ds = Epoch::chain_dataset_day(&day_bytes(d), ProgramClass::V4, 0, DATASET_LOG2); + let mh = ds.memhard().expect("memory-hard"); + assert_eq!(mh.params.shape.mixer_mult, 8); + assert_eq!(mh.params.shape.cache_log2_words, 26); + assert_eq!(mh.cache.line_mask(), (1u32 << 22) - 1); + log!( + "day {d}: day bytes {} key {} rot {:?} cache 2^26 words fnv {:016x} filled in {:.2} s", + hex(&day_bytes(d)), + mh.params.key.iter().map(|w| format!("{w:08x}")).collect::>().join(""), + mh.params.rot, + mh.cache.fnv1a64(), + t0.elapsed().as_secs_f64() + ); + for h in &per_round { + zero(h); + } + let next = AtomicUsize::new(0); + let mism = AtomicUsize::new(0); + let t1 = Instant::now(); + std::thread::scope(|sc| { + for _ in 0..threads { + let (next, mism, per_round, mp, cache) = (&next, &mism, &per_round, &mh.params, &mh.cache); + sc.spawn(move || { + let mut ts = [0u32; 64]; + let mut out = [[0u32; 16]; 64]; + let mut lines = [[0u32; 8]; 64]; + let mut lib = [[0u32; 16]; 64]; + loop { + let b = next.fetch_add(1, Ordering::Relaxed) as u64; + let start = b * batch; + if start >= items { + break; + } + for k in 0..64 { + ts[k] = (start + k as u64) as u32; + } + derive_traced(&ts, mp, cache, &mut out, &mut lines, plant.line_mask()); + for k in 0..64 { + for r in 0..ITEM_ROUNDS { + per_round[r][lines[k][r] as usize].fetch_add(1, Ordering::Relaxed); + } + } + let check = plant == Plant::None && (validate == "all" || (validate == "sample" && b % 64 == 0)); + if check { + igneum_pow::memhard::derive_items_leaves(&ts, mp, cache, Some(&state_leaves()), &mut lib); + for k in 0..64 { + if lib[k] != out[k] { + mism.fetch_add(1, Ordering::Relaxed); + } + } + } + } + }); + } + }); + let el = t1.elapsed().as_secs_f64(); + derivations += items; + let mm = mism.load(Ordering::Relaxed) as u64; + mismatches_total += mm; + log!( + "day {d}: {items} items ({} line reads) derived in {el:.1} s ({:.2} M items/s); library comparison: {} ({} mismatches)", + items * 8, + items as f64 / el / 1e6, + if plant == Plant::None { validate } else { "skipped under the plant" }, + mm + ); + if mm > 0 { + log!("day {d}: THE TRACED MIRROR DISAGREES WITH THE LIBRARY: {mm} items; nothing below is trusted"); + } + // per-day stats: per round, total over rounds, buckets of 64 lines (one segment) + let mut total = vec![0u64; lines_n]; + for r in 0..ITEM_ROUNDS { + let snap = snapshot(&per_round[r]); + let s = stats_of(snap.iter().copied(), lines_n); + log!("{}", s.line(&format!("day {d} round {r} lines"))); + for (i, c) in snap.iter().enumerate() { + total[i] += c; + pooled[r][i].fetch_add(*c as u32, Ordering::Relaxed); + } + } + let s_full = stats_of(total.iter().copied(), lines_n); + log!("{}", s_full.line(&format!("day {d} all rounds lines"))); + let b: Vec = total.chunks(64).map(|c| c.iter().sum()).collect(); + let s_b = stats_of(b.iter().copied(), b.len()); + log!("{}", s_b.line(&format!("day {d} all rounds buckets64"))); + let fire = six_sigma_test(&format!("day {d} buckets64"), &s_b); + any_fire |= fire; + // control of the same size + let ctrl = atomic_vec(lines_n); + control(lines_n, items * 8, 0xF8_0000 + d, threads, &ctrl); + let cs = snapshot(&ctrl); + let c_full = stats_of(cs.iter().copied(), lines_n); + log!("{}", c_full.line(&format!("day {d} CONTROL lines"))); + let cb: Vec = cs.chunks(64).map(|c| c.iter().sum()).collect(); + let c_b = stats_of(cb.iter().copied(), cb.len()); + log!("{}", c_b.line(&format!("day {d} CONTROL buckets64"))); + let hot = hot_set_test(&format!("day {d} lines"), &s_full, &c_full); + any_fire |= hot; + } + // pooled over the days + let mut total = vec![0u64; lines_n]; + for r in 0..ITEM_ROUNDS { + let snap = snapshot(&pooled[r]); + let s = stats_of(snap.iter().copied(), lines_n); + log!("{}", s.line(&format!("POOLED {days} days round {r} lines"))); + for (i, c) in snap.iter().enumerate() { + total[i] += c; + } + } + let s_full = stats_of(total.iter().copied(), lines_n); + log!("{}", s_full.line(&format!("POOLED {days} days all rounds lines"))); + let b: Vec = total.chunks(64).map(|c| c.iter().sum()).collect(); + let s_b = stats_of(b.iter().copied(), b.len()); + log!("{}", s_b.line(&format!("POOLED {days} days all rounds buckets64"))); + let fire = six_sigma_test(&format!("POOLED {days} days buckets64"), &s_b); + let fire_full = six_sigma_test(&format!("POOLED {days} days full 2^22 lines"), &s_full); + let ctrl = atomic_vec(lines_n); + control(lines_n, derivations * 8, 0xF8_1111, threads, &ctrl); + let cs = snapshot(&ctrl); + let c_full = stats_of(cs.iter().copied(), lines_n); + log!("{}", c_full.line(&format!("POOLED CONTROL lines"))); + let cb: Vec = cs.chunks(64).map(|c| c.iter().sum()).collect(); + let c_b = stats_of(cb.iter().copied(), cb.len()); + log!("{}", c_b.line(&format!("POOLED CONTROL buckets64"))); + let hot = hot_set_test(&format!("POOLED {days} days lines"), &s_full, &c_full); + any_fire |= fire | fire_full | hot; + // files + let tag = format!("lines-d{day0}-n{days}-i{items_log2}-{}", plant.name()); + { + let mut f = std::fs::File::create(format!("{out_dir}/{tag}-buckets64.txt")).unwrap(); + writeln!(f, "# bucket(64 lines = one segment) count ; pooled over {days} days from {day0}, items 2^{items_log2} per day, plant {}", plant.name()).unwrap(); + for (i, c) in b.iter().enumerate() { + writeln!(f, "{i} {c}").unwrap(); + } + let mut g = std::fs::File::create(format!("{out_dir}/{tag}-full.u32le")).unwrap(); + let mut bytes = Vec::with_capacity(lines_n * 4); + for c in &total { + bytes.extend_from_slice(&(*c as u32).to_le_bytes()); + } + g.write_all(&bytes).unwrap(); + } + log!( + "census lines DONE: {derivations} item derivations ({} line reads) over {days} days in {:.1} s; mirror mismatches {mismatches_total}; verdict {}", + derivations * 8, + t_all.elapsed().as_secs_f64(), + if any_fire { "FLAGGED" } else { "PASS (no test fired)" } + ); +} + +// -------------------------------------------------------------------------------------------------------------- +// Census 2 and 3: the warp interpreter mirrored with every item index recorded +// -------------------------------------------------------------------------------------------------------------- + +struct Table { + items: Vec<[u32; 16]>, + lines: Vec<[u32; 8]>, +} + +fn build_table(mp: &MixParams, cache: &Cache, threads: usize, plant: Plant, validate: &str) -> (Table, u64) { + let n = 1usize << (DATASET_LOG2 - 4); + let t0 = Instant::now(); + let mut items = vec![[0u32; 16]; n]; + let mut lines = vec![[0u32; 8]; n]; + let mism = AtomicUsize::new(0); + std::thread::scope(|sc| { + let per = n / threads; + for (th, (ic, lc)) in items.chunks_mut(per).zip(lines.chunks_mut(per)).enumerate() { + let mism = &mism; + sc.spawn(move || { + let mut ts = [0u32; 64]; + let mut lib = [[0u32; 16]; 64]; + let base = th * per; + for (bi, (ib, lb)) in ic.chunks_mut(64).zip(lc.chunks_mut(64)).enumerate() { + let start = base + bi * 64; + for k in 0..ib.len() { + ts[k] = (start + k) as u32; + } + derive_traced(&ts[..ib.len()], mp, cache, ib, lb, plant.line_mask()); + let check = plant == Plant::None && (validate == "all" || (validate == "sample" && bi % 64 == 0)); + if check { + igneum_pow::memhard::derive_items_leaves(&ts[..ib.len()], mp, cache, Some(&state_leaves()), &mut lib); + for k in 0..ib.len() { + if lib[k] != ib[k] { + mism.fetch_add(1, Ordering::Relaxed); + } + } + } + } + }); + } + }); + let mm = mism.load(Ordering::Relaxed) as u64; + log!( + "table: {n} items derived with their 8 lines in {:.1} s; library comparison {} ({mm} mismatches)", + t0.elapsed().as_secs_f64(), + if plant == Plant::None { validate } else { "skipped under the plant" } + ); + (Table { items, lines }, mm) +} + +struct ProgramSpec { + label: String, + epoch_seed: Vec, + era: Vec, +} + +fn program_spec(k: u32) -> ProgramSpec { + let genesis = unhex(GENESIS_HEX).unwrap(); + match k { + 1 => ProgramSpec { label: "p1-devnet-epoch0".into(), epoch_seed: genesis.clone(), era: genesis }, + _ => { + let w = |s: String| -> Vec { seed_words_from_bytes(s.as_bytes()).iter().flat_map(|x| x.to_le_bytes()).collect() }; + ProgramSpec { + label: format!("p{k}-attack-f8"), + epoch_seed: w(format!("igneum-attack-f8/program/{k}")), + era: w(format!("igneum-attack-f8/era/{k}")), + } + } + } +} + +/// The mirror of `verify::interpret_warp_init` for class v4 (the ops a v4 program can hold), reading dataset words +/// from the item table and reporting every load's item index. +/// The mirror of `verify::interpret_warp_init` for class v4 (the ops a v4 program can hold), reading dataset words +/// from the item table and reporting every load's item index and, per load position, how many lanes' source +/// register was saturated (0 or all ones) at the load. +struct Mirror<'a> { + program: &'a Program, + era: Option<&'a EraParams>, + layout: Layout, + mask: u32, + log2: u32, + table: &'a Table, + plant_site: Option, +} + +const POS: usize = 128; + +struct Sink { + /// the item index of every load position, per lane + items: Vec<[u32; LANES]>, + /// the masked word index (the address before the layout split) of every load position, per lane + addrs: Vec<[u32; LANES]>, + /// lanes whose source register read 0 or 2^32 - 1 at the load, per position + sat: [u32; POS], +} + +impl Sink { + fn new() -> Sink { + Sink { items: Vec::with_capacity(POS), addrs: Vec::with_capacity(POS), sat: [0; POS] } + } +} + +impl<'a> Mirror<'a> { + #[inline(always)] + fn step(&self, ins: &Instr, r: &mut [[u32; LANES]; 8], sel: &[u32; LANES], sink: &mut Sink, site: &mut usize) { + let d = ins.dst as usize; + let a = ins.src as usize; + match ins.op { + Op::Add => { + let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32); + let src = r[a]; + for lane in 0..LANES { + let s = (sel[lane] >> bit) & 1; + let c = if s != 0 { imm2 } else { imm }; + r[d][lane] = r[d][lane].wrapping_add(src[lane]).wrapping_add(c); + } + } + Op::Sub => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].wrapping_sub(src[lane]); + } + } + Op::Mul => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].wrapping_mul(src[lane]); + } + } + Op::MulHi => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = ((r[d][lane] as u64 * src[lane] as u64) >> 32) as u32; + } + } + Op::Xor => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] ^= src[lane]; + } + } + Op::Or => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] |= src[lane]; + } + } + Op::Rotl => { + let n = ins.rot; + for lane in 0..LANES { + r[d][lane] = r[d][lane].rotate_left(n); + } + } + Op::Rotr => { + let src = r[a]; + for lane in 0..LANES { + r[d][lane] = r[d][lane].rotate_right(src[lane] & 31); + } + } + Op::Mad => { + let src = r[a]; + let src2 = r[ins.src2 as usize]; + for lane in 0..LANES { + r[d][lane] = src[lane].wrapping_mul(src2[lane]).wrapping_add(r[d][lane]); + } + } + Op::Shfl => { + let src = r[a]; + let m = ins.mask as usize; + for lane in 0..LANES { + r[d][lane] ^= src[lane ^ m]; + } + } + Op::Load => { + assert_eq!(ins.width, 1, "class v4 loads one word"); + let mut ts = [0u32; LANES]; + let mut ws = [0u32; LANES]; + let p = sink.items.len(); + for lane in 0..LANES { + let x = r[a][lane]; + if x == 0 || x == u32::MAX { + sink.sat[p] += 1; + } + let idx = load_index(self.era, ins, x, self.mask, self.log2); + let (t, j) = self.layout.split(idx); + let t = if self.plant_site == Some(*site) { 0x00_1234 } else { t }; + ts[lane] = t; + ws[lane] = idx; + r[d][lane] ^= self.table.items[t as usize][j as usize]; + } + sink.items.push(ts); + sink.addrs.push(ws); + *site += 1; + } + Op::WLoad | Op::Scratch | Op::Hot => panic!("op {:?} is not a class v4 op", ins.op), + } + } + + /// Hashes of the warp at `base_nonce`; the sink carries the item index of every load (128 x 32). + fn warp(&self, base_nonce: u32, sink: &mut Sink) -> [u64; LANES] { + let seed = &self.program.seed; + let mut r = [[0u32; LANES]; 8]; + for lane in 0..LANES { + let nonce = base_nonce.wrapping_add(lane as u32); + for i in 0..8 { + let mut x = nonce ^ seed[i]; + x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1)); + x = splitmix32(x); + r[i][lane] = x ^ seed[(i + 1) & 7]; + } + } + sink.items.clear(); + sink.addrs.clear(); + sink.sat = [0; POS]; + for _ in 0..ITERATIONS { + let sel = r[0]; + let mut site = 0usize; + for ins in &self.program.instrs { + self.step(ins, &mut r, &sel, sink, &mut site); + } + for _ in 0..self.program.shadow_reps() { + for ins in &self.program.shadow { + self.step(ins, &mut r, &sel, sink, &mut site); + } + } + } + let mut hashes = [0u64; LANES]; + for lane in 0..LANES { + let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21); + let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27); + hashes[lane] = ((hi as u64) << 32) | lo as u64; + } + hashes + } +} + +fn pct(h: &[u64], q: f64) -> usize { + let total: u64 = h.iter().sum(); + let want = (total as f64 * q).ceil().max(1.0) as u64; + let mut acc = 0u64; + for (i, c) in h.iter().enumerate() { + acc += c; + if acc >= want { + return i; + } + } + h.len() - 1 +} + +fn dist_line(label: &str, h: &[u64]) -> String { + let total: u64 = h.iter().sum(); + let min = h.iter().position(|&c| c > 0).unwrap_or(0); + let max = h.iter().rposition(|&c| c > 0).unwrap_or(0); + let mean = h.iter().enumerate().map(|(i, c)| i as f64 * *c as f64).sum::() / total.max(1) as f64; + format!( + "{label}: n {total} min {min} p1 {} median {} p99 {} max {max} mean {mean:.4}", + pct(h, 0.01), + pct(h, 0.5), + pct(h, 0.99) + ) +} + +/// The window of every load site as an item range: `(first item, items)`; the era window is the top `k` bits of +/// the 28-bit word index and every layout position lies below 16, so the window is the same aligned range of items. +fn site_item_windows(program: &Program, mask: u32, log2: u32) -> Vec<(u32, u32)> { + program + .instrs + .iter() + .filter(|i| i.op == Op::Load) + .map(|i| { + let (wm, off) = igneum_pow::verify::window(i, mask, log2); + (off >> 4, (wm >> 4) + 1) + }) + .collect() +} + +/// The expected reads per item under the window layer alone (every site uniform on its own window), as a density per +/// quarter of the item space (windows are the dataset, a half or a quarter, aligned). +fn window_density(windows: &[(u32, u32)], items_n: usize, reads_per_site: f64) -> [f64; 4] { + let q = items_n as u32 / 4; + let mut d = [0f64; 4]; + for &(first, n) in windows { + for (k, dq) in d.iter_mut().enumerate() { + let qs = k as u32 * q; + if qs >= first && qs < first + n { + *dq += reads_per_site / n as f64; + } + } + } + d +} + +/// Statistics of `counts` against a per-quarter expected density: chi-square per dof, the largest and smallest +/// bucket in sigma of their own expectation, buckets of `per` items. +fn stats_vs_density(counts: &[u64], dens: &[f64; 4], per: usize) -> (f64, f64, u64, usize, f64) { + let items_n = counts.len(); + let q = items_n / 4; + let mut chi2 = 0f64; + let mut z_max = f64::MIN; + let mut z_min = f64::MAX; + let mut max = 0u64; + let mut argmax = 0usize; + let mut dof = 0usize; + for (b, c) in counts.chunks(per).enumerate() { + let e = dens[(b * per) / q] * per as f64; + if e <= 0.0 { + continue; + } + let s: u64 = c.iter().sum(); + let z = (s as f64 - e) / e.sqrt(); + chi2 += z * z; + dof += 1; + if z > z_max { + z_max = z; + max = s; + argmax = b; + } + if z < z_min { + z_min = z; + } + } + (chi2 / (dof.max(2) - 1) as f64, z_max, max, argmax, z_min) +} + +/// The windowed control: `n` reads, site `i mod 16`, uniform on the site's window. +fn windowed_control(windows: &[(u32, u32)], n: u64, seed: u64, threads: usize, hist: &[AtomicU32]) { + std::thread::scope(|sc| { + for th in 0..threads { + let hist = &hist; + sc.spawn(move || { + let mut s = SplitMix64::new(seed ^ (th as u64).wrapping_mul(0x9E3779B97F4A7C15)); + let per = n / threads as u64 + if (th as u64) < n % threads as u64 { 1 } else { 0 }; + for i in 0..per { + let (first, cnt) = windows[(i % 16) as usize]; + let t = first + (s.next() as u32 & (cnt - 1)); + hist[t as usize].fetch_add(1, Ordering::Relaxed); + } + }); + } + }); +} + +struct ProgramSummary { + label: String, + id: u64, + attempt: u32, + /// X_f against the windowed control at f = 0.1%, 0.5%, 1% + x: [f64; 3], + hot: bool, + /// windowed chi2/dof of the 64-item buckets and the largest bucket in sigma + chi2_w: f64, + z_w: f64, + /// the largest share of one hi16 bucket (256 items) at any load position + max_bucket_share: f64, + /// the largest saturated-source share at any load position + max_sat_share: f64, + /// acceptance-style 2,048-evaluation metrics: the largest count of one address at one position, the largest + /// saturated-source count at one position + acc_max_addr: u32, + acc_max_sat: u32, + /// S_0.1% over the window-model control and over the flat control + ratio_w: f64, + ratio_flat: f64, + /// top-f shares measured and under the window-model control, f = 0.1% and 1% + s01: f64, + e01: f64, + s10: f64, + e10: f64, + /// the hottest item, its count, and the lossy site whose saturated (or one-bit-off) source maps to it, if any + hottest: u32, + hottest_count: u64, + hottest_source: String, + /// share of each site's reads that land on the top 0.1% of items (the attribution pass; zeros without --diag) + by_site_share: [f64; 16], +} + +#[allow(clippy::too_many_arguments)] +fn run_program( + spec: &ProgramSpec, + ds: DatasetSource, + table: &Table, + table_mism: u64, + day: u64, + nonces: u64, + threads: usize, + plant: Plant, + check_every: u64, + diag: bool, + verbose: bool, + out_dir: &str, +) -> (ProgramSummary, DatasetSource) { + let program = Epoch::chain_program(&spec.epoch_seed, Some(&spec.era), ProgramClass::V5, &spec.label); + assert_eq!(program.generator, 5); + assert!(program.class.state, "class v5 keys the dataset by the state"); + assert_eq!(program.class.mixer_mult, 8); + assert_eq!(program.class.shadow.map(|s| (s.instrs, s.reps)), Some((256, 27))); + let era = program.class.era.expect("class v4 draws the era"); + let n_loads = program.instrs.iter().filter(|i| i.op == Op::Load).count(); + assert_eq!(n_loads, 16); + log!( + "program {}: epoch seed {} era seed {} id {:016x} attempt {} class {} op mix {}; era stride mul {:#010x} rot {} interleave {:?} windows {}", + spec.label, + hex(&spec.epoch_seed), + hex(&spec.era), + program.program_id(), + program.attempt, + program.class.name(), + program.op_mix(), + era.stride_mul, + era.stride_rot, + era.pos, + program.instrs.iter().enumerate().filter(|(_, i)| i.op == Op::Load).map(|(k, i)| format!("{k}:{}:{}", i.win, i.off)).collect::>().join(" ") + ); + let sites: Vec = program.instrs.iter().enumerate().filter(|(_, i)| i.op == Op::Load).map(|(k, _)| k).collect(); + if verbose { + // the static shape of every load site: its source register, the last base-program writer of that register + // before the site (cyclic) and whether that writer injects (add, sub, xor, mad, shfl, load), and how often the + // shadow block writes the register (the shadow runs between iteration i's instruction 63 and iteration i+1's + // instruction 0, outside the acceptance rule) + let instrs = &program.instrs; + for (k, ins) in instrs.iter().enumerate() { + if ins.op != Op::Load { + continue; + } + let src = ins.src; + let mut writer = String::from("none"); + let mut chain = Vec::new(); + for back in 1..instrs.len() { + let j = (k + instrs.len() - back) % instrs.len(); + if instrs[j].dst == src { + if writer == "none" { + writer = format!("{} at {j}", instrs[j].op.name()); + } + chain.push(format!("{}@{j}", instrs[j].op.name())); + if instrs[j].op.injects() { + break; + } + } + } + let mut sh: std::collections::BTreeMap<&str, usize> = std::collections::BTreeMap::new(); + for s in program.shadow.iter().filter(|s| s.dst == src) { + *sh.entry(s.op.name()).or_insert(0) += 1; + } + log!( + "load site instr {k}: src r{src} win {} off {}; last base writer {writer}; writers back to the last injecting one: {}; shadow writes of r{src} per rep: {}", + ins.win, + ins.off, + chain.join(" "), + sh.iter().map(|(o, n)| format!("{o}={n}")).collect::>().join(" ") + ); + } + } + let epoch = Epoch { program, dataset: ds }; + let plant_site = if plant == Plant::ConstItem { Some(0) } else { None }; + let mirror = Mirror { + program: &epoch.program, + era: epoch.program.class.era.as_ref(), + layout: epoch.program.class.layout(), + mask: epoch.dataset.mask, + log2: epoch.dataset.log2_words, + table, + plant_site, + }; + let items_n = 1usize << (DATASET_LOG2 - 4); + let windows = site_item_windows(&epoch.program, epoch.dataset.mask, epoch.dataset.log2_words); + let dens = window_density(&windows, items_n, nonces as f64 * 8.0); + log!( + "window layer: site item windows (first, items) {}; expected reads per item by quarter {:.3} {:.3} {:.3} {:.3} (flat uniform {:.3})", + windows.iter().map(|(f, n)| format!("({:#x},2^{})", f, n.trailing_zeros())).collect::>().join(" "), + dens[0], + dens[1], + dens[2], + dens[3], + nonces as f64 * 128.0 / items_n as f64 + ); + let item_hist = atomic_vec(items_n); + let pos_hi: Vec> = (0..POS).map(|_| atomic_vec(1 << 16)).collect(); + let pos_sat: Vec = atomic_vec(POS); + let warps = nonces / LANES as u64; + let next = AtomicUsize::new(0); + let checked = AtomicUsize::new(0); + let hash_mism = AtomicUsize::new(0); + let max_lines_hash = 8 * 128; + let max_lines_warp = max_lines_hash * LANES; + // the acceptance-style metric on the first 64 warps (2,048 evaluations): per position, the count of every + // masked address and of saturated sources + let acc = std::sync::Mutex::new((vec![std::collections::HashMap::::new(); POS], [0u32; POS])); + let t1 = Instant::now(); + let dists: Vec<(Vec, Vec, Vec, Vec)> = std::thread::scope(|sc| { + let mut hs = Vec::new(); + for _ in 0..threads { + let (next, checked, hash_mism, item_hist, mirror, epoch, pos_hi, pos_sat, acc) = + (&next, &checked, &hash_mism, &item_hist, &mirror, &epoch, &pos_hi, &pos_sat, &acc); + hs.push(sc.spawn(move || { + let mut lines_hash = vec![0u64; max_lines_hash + 1]; + let mut items_hash = vec![0u64; 129]; + let mut lines_warp = vec![0u64; max_lines_warp + 1]; + let mut items_warp = vec![0u64; 128 * LANES + 1]; + let mut sink = Sink::new(); + let mut lane_lines: Vec = Vec::with_capacity(max_lines_hash); + let mut lane_items: Vec = Vec::with_capacity(128); + let mut warp_lines: Vec = Vec::with_capacity(max_lines_warp); + let mut warp_items: Vec = Vec::with_capacity(128 * LANES); + loop { + let w = next.fetch_add(1, Ordering::Relaxed) as u64; + if w >= warps { + break; + } + let base = (w * LANES as u64) as u32; + let hashes = mirror.warp(base, &mut sink); + assert_eq!(sink.items.len(), POS); + if plant == Plant::None && (w < 64 || w % check_every == 0) { + let lib = epoch.hash_warp(base); + checked.fetch_add(1, Ordering::Relaxed); + if lib != hashes { + hash_mism.fetch_add(1, Ordering::Relaxed); + } + } + if w < 64 { + let mut g = acc.lock().unwrap(); + for p in 0..POS { + for lane in 0..LANES { + *g.0[p].entry(sink.addrs[p][lane]).or_insert(0) += 1; + } + g.1[p] += sink.sat[p]; + } + } + warp_lines.clear(); + warp_items.clear(); + for lane in 0..LANES { + lane_lines.clear(); + lane_items.clear(); + for load in &sink.items { + let t = load[lane]; + lane_items.push(t); + lane_lines.extend_from_slice(&mirror.table.lines[t as usize]); + } + warp_items.extend_from_slice(&lane_items); + warp_lines.extend_from_slice(&lane_lines); + lane_items.sort_unstable(); + lane_items.dedup(); + lane_lines.sort_unstable(); + lane_lines.dedup(); + items_hash[lane_items.len()] += 1; + lines_hash[lane_lines.len()] += 1; + } + for (p, load) in sink.items.iter().enumerate() { + for lane in 0..LANES { + let t = load[lane]; + item_hist[t as usize].fetch_add(1, Ordering::Relaxed); + pos_hi[p][(t >> 8) as usize].fetch_add(1, Ordering::Relaxed); + } + pos_sat[p].fetch_add(sink.sat[p], Ordering::Relaxed); + } + warp_items.sort_unstable(); + warp_items.dedup(); + warp_lines.sort_unstable(); + warp_lines.dedup(); + items_warp[warp_items.len()] += 1; + lines_warp[warp_lines.len()] += 1; + } + (lines_hash, items_hash, lines_warp, items_warp) + })); + } + hs.into_iter().map(|h| h.join().unwrap()).collect() + }); + let el = t1.elapsed().as_secs_f64(); + let mut lines_hash = vec![0u64; max_lines_hash + 1]; + let mut items_hash = vec![0u64; 129]; + let mut lines_warp = vec![0u64; max_lines_warp + 1]; + let mut items_warp = vec![0u64; 128 * LANES + 1]; + for (a, b, c, d) in &dists { + for (i, v) in a.iter().enumerate() { + lines_hash[i] += v; + } + for (i, v) in b.iter().enumerate() { + items_hash[i] += v; + } + for (i, v) in c.iter().enumerate() { + lines_warp[i] += v; + } + for (i, v) in d.iter().enumerate() { + items_warp[i] += v; + } + } + let ck = checked.load(Ordering::Relaxed); + let hm = hash_mism.load(Ordering::Relaxed); + log!( + "{} warps ({} nonces) interpreted in {el:.1} s ({:.3} ms per warp per thread); Epoch::hash_warp agreement on {ck} warps: {hm} mismatches{}", + warps, + warps * LANES as u64, + el * 1e3 * threads as f64 / warps as f64, + if plant != Plant::None { " (library comparison skipped under the plant)" } else { "" } + ); + if hm > 0 || table_mism > 0 { + log!("THE MIRROR DISAGREES WITH THE LIBRARY (hashes {hm}, items {table_mism}); nothing below is trusted"); + } + let tag = format!("warps-{}-d{day}-n{nonces}-{}", spec.label, plant.name()); + log!("{}", dist_line(&format!("{} distinct lines per hash", spec.label), &lines_hash)); + log!("{}", dist_line(&format!("{} distinct items per hash", spec.label), &items_hash)); + log!("{}", dist_line(&format!("{} distinct lines per warp", spec.label), &lines_warp)); + log!("{}", dist_line(&format!("{} distinct items per warp", spec.label), &items_warp)); + let exp = |k: f64, l: f64| l * (1.0 - (1.0 - 1.0 / l).powf(k)); + log!( + "uniform expectation: distinct lines per hash {:.3} of 1024 reads, per warp {:.1} of 32768 reads (2^22 lines); distinct items per hash {:.4} of 128, per warp {:.2} of 4096 (2^24 items)", + exp(1024.0, 4194304.0), + exp(32768.0, 4194304.0), + exp(128.0, 16777216.0), + exp(4096.0, 16777216.0) + ); + { + let mut f = std::fs::File::create(format!("{out_dir}/{tag}-distinct.txt")).unwrap(); + for (name, h) in [("lines_hash", &lines_hash), ("items_hash", &items_hash), ("lines_warp", &lines_warp), ("items_warp", &items_warp)] { + writeln!(f, "# distinct {name}: value count").unwrap(); + for (i, c) in h.iter().enumerate().filter(|(_, c)| **c > 0) { + writeln!(f, "{name} {i} {c}").unwrap(); + } + } + } + // per-position diagnostics: the largest hi16 bucket share and the saturated-source share + let mut max_bucket_share = 0f64; + let mut max_bucket_pos = 0usize; + let mut max_sat_share = 0f64; + let mut max_sat_pos = 0usize; + { + let mut f = std::fs::File::create(format!("{out_dir}/{tag}-positions.txt")).unwrap(); + writeln!(f, "# position iteration site instr max_hi16_bucket_share saturated_share").unwrap(); + for p in 0..POS { + let mx = pos_hi[p].iter().map(|x| x.load(Ordering::Relaxed)).max().unwrap_or(0) as f64 / nonces as f64; + let sat = pos_sat[p].load(Ordering::Relaxed) as f64 / nonces as f64; + writeln!(f, "{p} {} {} {} {:.6} {:.6}", p / 16, p % 16, sites[p % 16], mx, sat).unwrap(); + if mx > max_bucket_share { + max_bucket_share = mx; + max_bucket_pos = p; + } + if sat > max_sat_share { + max_sat_share = sat; + max_sat_pos = p; + } + } + } + log!( + "positions: largest hi16-bucket (256 items) share {:.4}% at p{} (iteration {}, site {}, instr {}); windowed expectation {:.4}%; largest saturated-source share {:.4}% at p{} (site {}, instr {})", + max_bucket_share * 100.0, + max_bucket_pos, + max_bucket_pos / 16, + max_bucket_pos % 16, + sites[max_bucket_pos % 16], + 100.0 * 256.0 / (windows[max_bucket_pos % 16].1 as f64), + max_sat_share * 100.0, + max_sat_pos, + max_sat_pos % 16, + sites[max_sat_pos % 16] + ); + // the acceptance-style metric over the first 2,048 evaluations + let (acc_max_addr, acc_max_sat, acc_pos, acc_sat_pos) = { + let g = acc.lock().unwrap(); + let mut ma = 0u32; + let mut mp = 0usize; + let mut ms = 0u32; + let mut msp = 0usize; + for p in 0..POS { + let m = g.0[p].values().copied().max().unwrap_or(0); + if m > ma { + ma = m; + mp = p; + } + if g.1[p] > ms { + ms = g.1[p]; + msp = p; + } + } + (ma, ms, mp, msp) + }; + log!( + "acceptance-style (2,048 evaluations, the rule's sample size): the most repeated address at one position {} of 2048 at p{} (site {}, instr {}); saturated sources at one position {} of 2048 at p{} (site {}, instr {}); uniform expectation: repeats 1 to 2, saturated 0", + acc_max_addr, + acc_pos, + acc_pos % 16, + sites[acc_pos % 16], + acc_max_sat, + acc_sat_pos, + acc_sat_pos % 16, + sites[acc_sat_pos % 16] + ); + // the cross-hash item histogram: flat statistics, the windowed null, the windowed control, the hot-set test + let snap = snapshot(&item_hist); + let s_items = stats_of(snap.iter().copied(), items_n); + log!("{}", s_items.line(&format!("{} item histogram (flat)", spec.label))); + let (chi2_w1, z_w1, max_w1, arg_w1, zmin_w1) = stats_vs_density(&snap, &dens, 1); + let (chi2_w, z_w, max_w, arg_w, zmin_w) = stats_vs_density(&snap, &dens, 64); + log!( + "{} item histogram against the window density: full 2^24 chi2/dof {:.5} largest {} (item {:#x}) at {:+.2} sigma smallest at {:+.2}; buckets64 chi2/dof {:.5} largest {} (bucket {}) at {:+.2} sigma smallest at {:+.2}", + spec.label, + chi2_w1, + max_w1, + arg_w1, + z_w1, + zmin_w1, + chi2_w, + max_w, + arg_w, + z_w, + zmin_w + ); + let total_reads = nonces * 128; + let ctrl = atomic_vec(items_n); + windowed_control(&windows, total_reads, 0xF8_3333 ^ program_hash(&spec.label), threads, &ctrl); + let cs = snapshot(&ctrl); + let c_items = stats_of(cs.iter().copied(), items_n); + let (c_chi2_w, c_z_w, _, _, c_zmin_w) = stats_vs_density(&cs, &dens, 64); + let (c_chi2_w1, c_z_w1, _, _, _) = stats_vs_density(&cs, &dens, 1); + log!("{}", c_items.line(&format!("{} WINDOWED CONTROL item histogram (flat)", spec.label))); + log!( + "{} WINDOWED CONTROL against the window density: full chi2/dof {:.5} largest {:+.2} sigma; buckets64 chi2/dof {:.5} largest {:+.2} smallest {:+.2}", + spec.label, + c_chi2_w1, + c_z_w1, + c_chi2_w, + c_z_w, + c_zmin_w + ); + let f1 = z_w > 6.0 || zmin_w < -6.0; + log!( + "6-sigma {} items buckets64 against the window density: largest bucket {:+.2} sigma, smallest {:+.2} -> {}", + spec.label, + z_w, + zmin_w, + if f1 { "FLAGGED (beyond 6 sigma)" } else { "within 6 sigma" } + ); + let f3 = hot_set_test(&format!("{} items (windowed control)", spec.label), &s_items, &c_items); + let flat = atomic_vec(items_n); + control(items_n, total_reads, 0xF8_4444 ^ program_hash(&spec.label), threads, &flat); + let fs = snapshot(&flat); + let f_items = stats_of(fs.iter().copied(), items_n); + log!("{}", f_items.line(&format!("{} FLAT CONTROL item histogram", spec.label))); + let _ = hot_set_test(&format!("{} items (flat control, the auditor's first view)", spec.label), &s_items, &f_items); + let mut x = [0f64; 3]; + let mut ratio_w = [0f64; 3]; + let mut ratio_flat = [0f64; 3]; + for k in 0..3 { + x[k] = s_items.top[k].1 - c_items.top[k].1; + ratio_w[k] = s_items.top[k].1 / c_items.top[k].1; + ratio_flat[k] = s_items.top[k].1 / f_items.top[k].1; + } + log!( + "ratio {}: top 0.1% / 0.5% / 1% share over the WINDOW-MODEL control {:.4}x / {:.4}x / {:.4}x (gate 1.2x at 0.1%: {}); over the FLAT control {:.4}x / {:.4}x / {:.4}x", + spec.label, + ratio_w[0], + ratio_w[1], + ratio_w[2], + if ratio_w[0] <= 1.2 { "within" } else { "BEYOND" }, + ratio_flat[0], + ratio_flat[1], + ratio_flat[2] + ); + { + let mut g = std::fs::File::create(format!("{out_dir}/{tag}-items.u32le")).unwrap(); + let mut bytes = Vec::with_capacity(items_n * 4); + for c in &snap { + bytes.extend_from_slice(&(*c as u32).to_le_bytes()); + } + g.write_all(&bytes).unwrap(); + } + let mut by_site_share = [0f64; 16]; + if diag { + // attribution: the hottest items (the top 0.1% by count, and the top 8) traced back to the load positions + let want = (items_n as f64 / 1000.0).round() as u64; + let mut coc: std::collections::BTreeMap = std::collections::BTreeMap::new(); + for &c in &snap { + *coc.entry(c).or_insert(0) += 1; + } + let mut left = want; + let mut thr = 0u64; + for (&c, &n) in coc.iter().rev() { + thr = c; + if n >= left { + break; + } + left -= n; + } + let mut hot = vec![0u8; items_n]; + let mut n_hot = 0u64; + let mut hot_reads = 0u64; + for (t, &c) in snap.iter().enumerate() { + if c >= thr { + hot[t] = 1; + n_hot += 1; + hot_reads += c; + } + } + let mut top8: Vec<(u64, usize)> = snap.iter().enumerate().map(|(t, &c)| (c, t)).collect(); + top8.sort_unstable_by(|a, b| b.cmp(a)); + top8.truncate(8); + log!( + "attribution: hot threshold count >= {thr} marks {n_hot} items ({:.4}% of items) holding {hot_reads} reads ({:.4}% of reads); top 8 items {}", + n_hot as f64 * 100.0 / items_n as f64, + hot_reads as f64 * 100.0 / total_reads as f64, + top8.iter().map(|(c, t)| format!("t={t:#08x}:{c}")).collect::>().join(" ") + ); + let per_pos: Vec = (0..POS).map(|_| std::sync::atomic::AtomicU64::new(0)).collect(); + let per_top: Vec> = (0..8).map(|_| (0..POS).map(|_| std::sync::atomic::AtomicU64::new(0)).collect()).collect(); + let next2 = AtomicUsize::new(0); + std::thread::scope(|sc| { + for _ in 0..threads { + let (next2, mirror, hot, top8, per_pos, per_top) = (&next2, &mirror, &hot, &top8, &per_pos, &per_top); + sc.spawn(move || { + let mut sink = Sink::new(); + loop { + let w = next2.fetch_add(1, Ordering::Relaxed) as u64; + if w >= warps { + break; + } + mirror.warp((w * LANES as u64) as u32, &mut sink); + for (p, load) in sink.items.iter().enumerate() { + for lane in 0..LANES { + let t = load[lane] as usize; + if hot[t] != 0 { + per_pos[p].fetch_add(1, Ordering::Relaxed); + } + for (k, (_, tt)) in top8.iter().enumerate() { + if *tt == t { + per_top[k][p].fetch_add(1, Ordering::Relaxed); + } + } + } + } + } + }); + } + }); + let expect = n_hot as f64 / items_n as f64; + let mut rows: Vec<(usize, u64)> = per_pos.iter().enumerate().map(|(p, c)| (p, c.load(Ordering::Relaxed))).collect(); + rows.sort_unstable_by(|a, b| b.1.cmp(&a.1)); + log!( + "attribution: reads on hot items per position (flat expectation {:.4}% of each position's {nonces} reads); the 24 largest: {}", + expect * 100.0, + rows.iter().take(24).map(|(p, c)| format!("p{p}(it{} s{})={:.3}%", p / 16, p % 16, *c as f64 * 100.0 / nonces as f64)).collect::>().join(" ") + ); + let mut by_iter = vec![0u64; ITERATIONS]; + let mut by_site = vec![0u64; 16]; + for (p, c) in &rows { + by_iter[p / 16] += c; + by_site[p % 16] += c; + } + for s in 0..16 { + by_site_share[s] = by_site[s] as f64 / (nonces * 8) as f64; + } + log!( + "attribution: hot reads by iteration {} ; by site {}", + by_iter.iter().enumerate().map(|(i, c)| format!("it{i}={:.3}%", *c as f64 * 100.0 / (nonces * 16) as f64)).collect::>().join(" "), + by_site.iter().enumerate().map(|(i, c)| format!("s{i}={:.3}%", *c as f64 * 100.0 / (nonces * 8) as f64)).collect::>().join(" ") + ); + // per-site contribution table: window, share of the site's reads into the top 0.1% set, the entropy of the + // site's item index over nonces (hi16 buckets of 256 items: a window of 2^(24 - k_off) items is 2^(16 - k_off) + // buckets, so uniform on the window = 16 - k_off bits; the runs of 09:04 UTC printed 16 - 2 k_off by mistake), + // saturated-source share, largest bucket share + let load_instrs: Vec<&Instr> = epoch.program.instrs.iter().filter(|i| i.op == Op::Load).collect(); + for s in 0..16 { + let mut acc = vec![0u64; 1 << 16]; + let mut hot_s = 0u64; + let mut sat_s = 0u64; + for it in 0..ITERATIONS { + let p = it * 16 + s; + for (b, c) in pos_hi[p].iter().enumerate() { + acc[b] += c.load(Ordering::Relaxed) as u64; + } + hot_s += per_pos[p].load(Ordering::Relaxed); + sat_s += pos_sat[p].load(Ordering::Relaxed) as u64; + } + let n_s = (nonces * ITERATIONS as u64) as f64; + let mut h = 0f64; + let mut mx = 0u64; + for &c in &acc { + if c > 0 { + let pr = c as f64 / n_s; + h -= pr * pr.log2(); + } + mx = mx.max(c); + } + let ins = load_instrs[s]; + log!( + "site {s} (instr {}, src r{}, k_off {} offset {}, window 2^{} items): share of its reads into the top 0.1% {:.4}% (flat expectation {:.4}%); index entropy {:.3} bits of {} uniform-on-window; saturated source {:.4}%; largest 256-item bucket {:.4}% (window expectation {:.4}%)", + sites[s], + ins.src, + ins.win, + ins.off, + windows[s].1.trailing_zeros(), + hot_s as f64 * 100.0 / n_s, + expect * 100.0, + h, + 16 - ins.win.min(2) as u32, + sat_s as f64 * 100.0 / n_s, + mx as f64 * 100.0 / n_s, + 100.0 * 256.0 / windows[s].1 as f64 + ); + } + for (k, (c, t)) in top8.iter().enumerate() { + let mut pos: Vec<(usize, u64)> = per_top[k].iter().enumerate().map(|(p, x)| (p, x.load(Ordering::Relaxed))).filter(|(_, x)| *x > 0).collect(); + pos.sort_unstable_by(|a, b| b.1.cmp(&a.1)); + log!( + "attribution: item {t:#08x} ({c} reads) read from positions {}", + pos.iter().take(12).map(|(p, x)| format!("p{p}(it{} s{})x{x}", p / 16, p % 16)).collect::>().join(" ") + ); + } + } + // the hottest item and the lossy site whose saturated source maps to it under the era map (all ones, zero, or + // one bit off either), with that site's source register and its last base-program writer + let (hottest, hottest_count) = snap.iter().enumerate().map(|(t, &c)| (t as u32, c)).max_by_key(|&(_, c)| c).unwrap(); + let hottest_source = { + let instrs = &epoch.program.instrs; + let layout = epoch.program.class.layout(); + let era = epoch.program.class.era.as_ref(); + let mut found = String::from("none"); + for (k, ins) in instrs.iter().enumerate() { + if ins.op != Op::Load { + continue; + } + let img = |x: u32| layout.split(load_index(era, ins, x, epoch.dataset.mask, epoch.dataset.log2_words)).0; + let mut cands: Vec<(u32, &str)> = vec![(u32::MAX, "all-ones"), (0, "zero")]; + for b in 0..32 { + cands.push((u32::MAX ^ (1 << b), "one-zero-bit")); + cands.push((1 << b, "one-one-bit")); + } + if let Some((_, what)) = cands.iter().find(|(x, _)| img(*x) == hottest) { + let mut writer = String::from("none"); + for back in 1..instrs.len() { + let j = (k + instrs.len() - back) % instrs.len(); + if instrs[j].dst == ins.src { + writer = format!("{}@{j}", instrs[j].op.name()); + break; + } + } + found = format!("site-instr{k}:r{}:{what}:last-writer-{writer}", ins.src); + break; + } + } + found + }; + log!( + "hottest item {} ({:#08x}): {} reads ({:.4}% of all); predicted source {}", + spec.label, + hottest, + hottest_count, + hottest_count as f64 * 100.0 / total_reads as f64, + hottest_source + ); + let hot = f3; + log!( + "program {} DONE: {} nonces; 6-sigma (windowed) {}; hot-set {}; verdict {}", + spec.label, + nonces, + if f1 { "FLAGGED" } else { "clear" }, + if hot { "FLAGGED" } else { "clear" }, + if f1 || hot { "FLAGGED" } else { "PASS (no test fired)" } + ); + let summary = ProgramSummary { + label: spec.label.clone(), + id: epoch.program.program_id(), + attempt: epoch.program.attempt, + x, + hot, + chi2_w, + z_w, + max_bucket_share, + max_sat_share, + acc_max_addr, + acc_max_sat, + ratio_w: ratio_w[0], + ratio_flat: ratio_flat[0], + s01: s_items.top[0].1, + e01: c_items.top[0].1, + s10: s_items.top[2].1, + e10: c_items.top[2].1, + hottest, + hottest_count, + hottest_source, + by_site_share, + }; + (summary, epoch.dataset) +} + +fn program_hash(label: &str) -> u64 { + igneum_pow::seed::fnv1a64(label.as_bytes()) +} + +#[allow(clippy::too_many_arguments)] +fn census_warps(programs: &[u32], day: u64, nonces: u64, threads: usize, plant: Plant, validate: &str, check_every: u64, diag: bool, out_dir: &str) { + log!( + "census warps: programs {:?} day {day} nonces {nonces} threads {threads} plant {} validate {validate} check_every {check_every} diag {diag}", + programs, + plant.name() + ); + let t0 = Instant::now(); + let mut ds: DatasetSource = Epoch::chain_dataset_day(&day_bytes(day), ProgramClass::V5, 0, DATASET_LOG2).with_leaves(state_leaves()); + assert_eq!(ds.log2_words, DATASET_LOG2); + log!("class v5: state leaves {} (fnv {:016x}), root {}", state_leaves().n(), state_leaves().fnv1a64(), hex(&state_leaves().root)); + let (table, table_mism) = { + let mh = ds.memhard().unwrap(); + log!("day {day}: cache filled in {:.2} s, fnv {:016x}", t0.elapsed().as_secs_f64(), mh.cache.fnv1a64()); + build_table(&mh.params, &mh.cache, threads, plant, validate) + }; + let verbose = programs.len() <= 3; + let mut summaries = Vec::new(); + for &k in programs { + let spec = program_spec(k); + let (s, d) = run_program(&spec, ds, &table, table_mism, day, nonces, threads, plant, check_every, diag, verbose, out_dir); + ds = d; + summaries.push(s); + } + if programs.len() > 1 { + // the re-gate format (Counter ASIC lane, 7 October 2026): one line per seed, then PASS or FAIL against 1.2x + let mut over = Vec::new(); + for s in &summaries { + log!( + "SEED {} id {:016x} attempt {} S0.1 {:.5}% null0.1 {:.5}% ratio {:.4}x S1.0 {:.5}% null1.0 {:.5}% hottest {:#08x}:{} source {} by-site {}", + s.label, + s.id, + s.attempt, + s.s01 * 100.0, + s.e01 * 100.0, + s.ratio_w, + s.s10 * 100.0, + s.e10 * 100.0, + s.hottest, + s.hottest_count, + s.hottest_source, + if diag { s.by_site_share.iter().enumerate().map(|(i, x)| format!("s{i}={:.3}%", x * 100.0)).collect::>().join(" ") } else { "(no --diag)".to_string() } + ); + if s.ratio_w > 1.2 { + over.push(format!("{}({:.3}x)", s.label, s.ratio_w)); + } + } + log!( + "CENSUS {}: {} seeds at {} nonces, {} over 1.2x of the window model at the top 0.1%{}{}", + if over.is_empty() { "PASS" } else { "FAIL" }, + summaries.len(), + nonces, + over.len(), + if over.is_empty() { "" } else { ": " }, + over.join(" ") + ); + let mut f = std::fs::File::create(format!("{out_dir}/seed-census-d{day}-n{nonces}-p{}-{}.txt", programs[0], programs[programs.len() - 1])).unwrap(); + writeln!(f, "# label id attempt X_0.1% X_0.5% X_1% hot chi2w_buckets64 z_w max_hi16_bucket_share max_sat_share acc_max_addr_of_2048 acc_max_sat_of_2048 ratio_0.1%_window ratio_0.1%_flat").unwrap(); + let mut n_hot = 0usize; + let mut n_acc_addr = [0usize; 3]; + let mut n_acc_sat = [0usize; 3]; + let mut n_sixsig = 0usize; + let mut n_ratio = 0usize; + for s in &summaries { + n_ratio += (s.ratio_w > 1.2) as usize; + writeln!( + f, + "{} {:016x} {} {:.5} {:.5} {:.5} {} {:.4} {:.2} {:.6} {:.6} {} {} {:.4} {:.4}", + s.label, s.id, s.attempt, s.x[0] * 100.0, s.x[1] * 100.0, s.x[2] * 100.0, s.hot as u8, s.chi2_w, s.z_w, s.max_bucket_share, s.max_sat_share, s.acc_max_addr, s.acc_max_sat, s.ratio_w, s.ratio_flat + ) + .unwrap(); + n_hot += s.hot as usize; + n_sixsig += (s.z_w > 6.0) as usize; + for (i, thr) in [3u32, 11, 21].into_iter().enumerate() { + n_acc_addr[i] += (s.acc_max_addr >= thr) as usize; + n_acc_sat[i] += (s.acc_max_sat >= thr) as usize; + } + } + log!( + "SEED CENSUS: {} programs at {} nonces each: hot set (X_f >= f, windowed control) in {}; top 0.1% beyond 1.2x of the window-model control in {}; windowed buckets64 beyond 6 sigma in {}; acceptance-style most-repeated address at one position >= 3 / 11 / 21 of 2048 in {} / {} / {}; saturated sources at one position >= 3 / 11 / 21 of 2048 in {} / {} / {}", + summaries.len(), + nonces, + n_hot, + n_ratio, + n_sixsig, + n_acc_addr[0], + n_acc_addr[1], + n_acc_addr[2], + n_acc_sat[0], + n_acc_sat[1], + n_acc_sat[2] + ); + let mut by_x: Vec<&ProgramSummary> = summaries.iter().collect(); + by_x.sort_by(|a, b| b.x[0].partial_cmp(&a.x[0]).unwrap()); + for s in by_x.iter().take(10) { + log!( + "SEED CENSUS top: {} id {:016x} attempt {} ratio_0.1% window {:.3}x flat {:.3}x X_0.1% {:+.4}% X_1% {:+.4}% hot {} z_w {:+.1} max bucket share {:.4}% max sat share {:.4}% acc addr {} acc sat {}", + s.label, + s.id, + s.attempt, + s.ratio_w, + s.ratio_flat, + s.x[0] * 100.0, + s.x[2] * 100.0, + s.hot as u8, + s.z_w, + s.max_bucket_share * 100.0, + s.max_sat_share * 100.0, + s.acc_max_addr, + s.acc_max_sat + ); + } + } +} +fn usage() -> ! { + eprintln!( + "attack-f8 lines --day0 20730 --days 16 --items-log2 24 --threads 12 --out [--plant none|quarter-lines|half-lines] [--validate all|sample|none]\n\ + attack-f8 warps --program 1|2|3 | --programs a..b --day 20730 --nonces 1000000 --threads 12 --out [--plant none|const-item] [--validate all|sample|none] [--check-every 997] [--diag 1]\n\ + attack-f8 census --seeds 64 --nonces 16777216 --control window --by-site --threads 12 --out the re-gate: one line per seed, PASS/FAIL against 1.2x\n\ + attack-f8 static --programs 1..67 per program, every load site's write chain back to the last injecting write (rule (a'))\n\ + attack-f8 seeds print the three programs' seeds" + ); + std::process::exit(2); +} + +fn main() { + let args: Vec = std::env::args().collect(); + if args.len() < 2 { + usage(); + } + let get = |k: &str, d: &str| -> String { + let mut i = 2; + while i + 1 < args.len() { + if args[i] == k { + return args[i + 1].clone(); + } + i += 1; + } + d.to_string() + }; + let threads: usize = get("--threads", "12").parse().unwrap(); + let state_path = get("--state", ""); + if !state_path.is_empty() { + let stream = StateStream::read_file(std::path::Path::new(&state_path)).expect("state stream"); + let leaves = StateLeaves::from_stream(&stream, DATASET_LOG2); + log!("state stream {}: block {} number {} root {} records {} -> {} leaves fnv {:016x}", state_path, hex(&stream.block), stream.number, hex(&stream.root), stream.records.len(), leaves.n(), leaves.fnv1a64()); + STATE.set(Arc::new(leaves)).ok(); + } + let out = get("--out", "."); + let plant = Plant::parse(&get("--plant", "none")); + let validate = get("--validate", "all"); + log!("adv-live-v5 {} (igneum-pow class v5, generator {}); args {:?}", env!("CARGO_PKG_VERSION"), igneum_pow::generator::GENERATOR_VERSION_V5, &args[1..]); + match args[1].as_str() { + "lines" => census_lines( + get("--day0", &DEFAULT_DAY.to_string()).parse().unwrap(), + get("--days", "16").parse().unwrap(), + get("--items-log2", "24").parse().unwrap(), + threads, + plant, + &validate, + &out, + ), + "warps" => census_warps( + &{ + let ps = get("--programs", ""); + if ps.is_empty() { + vec![get("--program", "1").parse::().unwrap()] + } else { + let (a, b) = ps.split_once("..").expect("--programs a..b"); + (a.parse::().unwrap()..=b.parse::().unwrap()).collect::>() + } + }, + get("--day", &DEFAULT_DAY.to_string()).parse().unwrap(), + get("--nonces", "1000000").parse().unwrap(), + threads, + plant, + &validate, + get("--check-every", "997").parse().unwrap(), + get("--diag", "0") == "1", + &out, + ), + "census" => { + // the re-gate entry: `census --seeds 64 --nonces 16777216 --control window --by-site` runs programs 2..=65 + // (the chain-shaped tag seeds; `--programs a..b` overrides) with the window-model control (the only control + // the hot-set test uses; the flat control is printed beside it) and the by-site attribution + let ps = get("--programs", ""); + let programs: Vec = if ps.is_empty() { + let n: u32 = get("--seeds", "64").parse().unwrap(); + (2..=n + 1).collect() + } else { + let (a, b) = ps.split_once("..").expect("--programs a..b"); + (a.parse::().unwrap()..=b.parse::().unwrap()).collect() + }; + let by_site = args.iter().any(|a| a == "--by-site") || get("--diag", "0") == "1"; + census_warps( + &programs, + get("--day", &DEFAULT_DAY.to_string()).parse().unwrap(), + get("--nonces", "16777216").parse().unwrap(), + threads, + plant, + &validate, + get("--check-every", "997").parse().unwrap(), + by_site, + &out, + ); + } + "static" => { + // per program: for every load site, the ops that write its source register between the register's last + // injecting write (add, sub, xor, mad, shfl, load) and the load, walking back cyclically; the proposed + // rule (a') rejects a program with an `or` or a `mul` in any such chain + let ps = get("--programs", "1..3"); + let (a, b) = ps.split_once("..").expect("--programs a..b"); + println!("# label id attempt sites_with_or sites_with_mul sites_with_mulhi sites_with_rot_only sites_clean rule_a_prime chains"); + for k in a.parse::().unwrap()..=b.parse::().unwrap() { + let spec = program_spec(k); + let program = Epoch::chain_program(&spec.epoch_seed, Some(&spec.era), ProgramClass::V4, &spec.label); + let instrs = &program.instrs; + let (mut n_or, mut n_mul, mut n_mulhi, mut n_rot, mut n_clean) = (0, 0, 0, 0, 0); + let mut chains = Vec::new(); + for (k, ins) in instrs.iter().enumerate() { + if ins.op != Op::Load { + continue; + } + let mut ops = Vec::new(); + for back in 1..instrs.len() { + let j = (k + instrs.len() - back) % instrs.len(); + if instrs[j].dst == ins.src { + if instrs[j].op.injects() { + ops.push(format!("{}@{j}", instrs[j].op.name())); + break; + } + ops.push(format!("{}@{j}", instrs[j].op.name())); + } + } + let has = |o: Op| ops.iter().any(|x| x.starts_with(&format!("{}@", o.name()))); + if has(Op::Or) { + n_or += 1; + } else if has(Op::Mul) { + n_mul += 1; + } else if has(Op::MulHi) { + n_mulhi += 1; + } else if has(Op::Rotl) || has(Op::Rotr) { + n_rot += 1; + } else { + n_clean += 1; + } + // the items the saturated sources map to under the era map at the 2^28-word dataset + let mask = (1u32 << DATASET_LOG2) - 1; + let layout = program.class.layout(); + let img = |x: u32| layout.split(load_index(program.class.era.as_ref(), ins, x, mask, DATASET_LOG2)).0; + chains.push(format!("{k}:r{}:{}:ones={:#08x}:zero={:#08x}", ins.src, ops.join("<"), img(u32::MAX), img(0))); + } + println!( + "{} {:016x} {} {n_or} {n_mul} {n_mulhi} {n_rot} {n_clean} {} {}", + spec.label, + program.program_id(), + program.attempt, + if n_or + n_mul > 0 { "REJECT" } else { "accept" }, + chains.join(" ") + ); + } + } + "seeds" => { + for k in 1..=3 { + let s = program_spec(k); + println!("{} epoch_seed {} era {}", s.label, hex(&s.epoch_seed), hex(&s.era)); + } + } + _ => usage(), + } +} diff --git a/tools/attack/adv-accept/run-box.sh b/tools/attack/adv-accept/run-box.sh index 125513995..68981fbac 100644 --- a/tools/attack/adv-accept/run-box.sh +++ b/tools/attack/adv-accept/run-box.sh @@ -12,8 +12,11 @@ dir=/srv/builds/_adv-adv-accept # OUTSIDE the worktree mirror (correction 18:5 mkdir -p "$dir" log="$dir/$tag.log" [ -x "$bin" ] || { echo "no binary at $bin" >&2; exit 1; } -setsid nice -n 10 taskset -c 8-95 "$bin" "$@" > "$log" 2>&1 < /dev/null & +# the per-box sweep lock (coordinator, 19:55 BST, 7 October 2026): one bounded run per box at a time, the rest wait in +# order; the lock holder is the recorded pid (kill -- - ends the wait or the run) +printf -v cmd '%q ' nice -n 10 taskset -c 8-95 "$bin" "$@" +setsid flock /srv/builds/_adv/locks/sweep.lock -c "$cmd > $(printf '%q' "$log") 2>&1" < /dev/null > /dev/null 2>&1 & pid=$! pgid=$(ps -o pgid= -p "$pid" 2>/dev/null | tr -d ' '); [ -n "$pgid" ] || pgid="$pid" echo "$pid" > "$log.pid"; echo "$pgid" > "$log.pgid" -echo "$(date -u +%FT%TZ) started pid $pid pgid $pgid nice 10 cores 8-95; log $log" | tee -a "$log.start" +echo "$(date -u +%FT%TZ) queued pid $pid pgid $pgid behind the per-box sweep lock; nice 10 cores 8-95; log $log" | tee -a "$log.start"