adv-accept: class-v5 variant of the live harness (vendored class-v5 igneum-pow, the Devnet 3 v5 pack's state stream); run-box.sh takes the per-box sweep lock; rule change 19:55 BST recorded

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
igneum-labs 2026-10-07 18:57:38 +00:00
parent 8a290dee72
commit 596a2f0d03
32 changed files with 15846 additions and 2 deletions

View file

@ -255,6 +255,13 @@ it folds in act as fresh randomness on both sides: the closed form and the live
the 50-program widening is running (gap-50.log). Also: the cheap 256-unit proxy did rank real tail programs,
and 100064 sits 0.0032 above the 0.98 floor at 2^20, so the floor is live in the population tail, not idle.
### Rule change 19:55 BST (box-hours honesty)
The bounded class became 88 cores per box in total across all lanes: no new sweep starts except under the
per-box lock `flock /srv/builds/_adv/locks/sweep.lock`, one sweep per box at a time; the running shards 00
and 01 finish as they are; my adv-live confirmations and the row-90 census count against the ceiling, so
nothing new starts until they end. Shards 02 to 09, the class-v5 check and any re-run go through the lock.
### Live hot-set census (RUNNING)
- box 2, 19:5x BST: `adv-live census --programs 2..201 --nonces 1000000 --threads 32` (200 consecutive accepted

112
tools/attack/adv-accept-v5/Cargo.lock generated Normal file
View file

@ -0,0 +1,112 @@
# This file is automatically @generated by Cargo.
# It is not intended for manual editing.
version = 4
[[package]]
name = "adv-accept-v5"
version = "0.1.0"
dependencies = [
"igneum-pow",
]
[[package]]
name = "igneum-pow"
version = "0.2.0"
dependencies = [
"serde_json",
]
[[package]]
name = "itoa"
version = "1.0.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
[[package]]
name = "memchr"
version = "2.8.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
[[package]]
name = "proc-macro2"
version = "1.0.107"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
dependencies = [
"unicode-ident",
]
[[package]]
name = "quote"
version = "1.0.47"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
dependencies = [
"proc-macro2",
]
[[package]]
name = "serde"
version = "1.0.229"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
dependencies = [
"serde_core",
]
[[package]]
name = "serde_core"
version = "1.0.229"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
dependencies = [
"serde_derive",
]
[[package]]
name = "serde_derive"
version = "1.0.229"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
dependencies = [
"proc-macro2",
"quote",
"syn",
]
[[package]]
name = "serde_json"
version = "1.0.151"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
dependencies = [
"itoa",
"memchr",
"serde",
"serde_core",
"zmij",
]
[[package]]
name = "syn"
version = "3.0.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
dependencies = [
"proc-macro2",
"quote",
"unicode-ident",
]
[[package]]
name = "unicode-ident"
version = "1.0.26"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
[[package]]
name = "zmij"
version = "1.0.23"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"

View file

@ -0,0 +1,21 @@
[package]
name = "adv-accept-v5"
version = "0.1.0"
edition = "2021"
description = "adv-accept lane: the f8 hot-set harness over class v5 (vendored igneum-pow of branch class-v5, the dataset keyed by the window's state leaves), to ask whether a class v4 finding still reads hot under class v5"
license = "MIT"
publish = false
[[bin]]
name = "adv-live-v5"
path = "src/live.rs"
[dependencies]
igneum-pow = { path = "igneum-pow" }
[workspace]
[profile.release]
opt-level = 3
lto = true
codegen-units = 1

View file

@ -0,0 +1 @@
target/

View file

@ -0,0 +1,105 @@
# This file is automatically @generated by Cargo.
# It is not intended for manual editing.
version = 4
[[package]]
name = "igneum-pow"
version = "0.2.0"
dependencies = [
"serde_json",
]
[[package]]
name = "itoa"
version = "1.0.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
[[package]]
name = "memchr"
version = "2.8.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
[[package]]
name = "proc-macro2"
version = "1.0.107"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
dependencies = [
"unicode-ident",
]
[[package]]
name = "quote"
version = "1.0.47"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
dependencies = [
"proc-macro2",
]
[[package]]
name = "serde"
version = "1.0.229"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
dependencies = [
"serde_core",
]
[[package]]
name = "serde_core"
version = "1.0.229"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
dependencies = [
"serde_derive",
]
[[package]]
name = "serde_derive"
version = "1.0.229"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
dependencies = [
"proc-macro2",
"quote",
"syn",
]
[[package]]
name = "serde_json"
version = "1.0.151"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
dependencies = [
"itoa",
"memchr",
"serde",
"serde_core",
"zmij",
]
[[package]]
name = "syn"
version = "3.0.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
dependencies = [
"proc-macro2",
"quote",
"unicode-ident",
]
[[package]]
name = "unicode-ident"
version = "1.0.26"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
[[package]]
name = "zmij"
version = "1.0.23"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"

View file

@ -0,0 +1,30 @@
[package]
name = "igneum-pow"
version = "0.2.0"
edition = "2021"
description = "Igneum random-program GPU proof-of-work: seed, program generator (version 2: fixed load count, fresh sources, acceptance rule), memory-hard dataset, CPU warp verifier and kernel emitters; the source of every program pack"
license = "MIT"
publish = false
[lib]
name = "igneum_pow"
path = "src/lib.rs"
[[bin]]
name = "igneum-pow"
path = "src/main.rs"
[dependencies]
[dev-dependencies]
serde_json = "1"
# The cache fill is 2^22 ChaCha12 blocks and the vector tests derive thousands of items.
# Unoptimised builds would make `cargo test` take minutes, so the dev profile is optimised too.
[profile.dev]
opt-level = 3
[profile.release]
opt-level = 3
lto = true
codegen-units = 1

View file

@ -0,0 +1,181 @@
# igneum-pow
The Igneum lottery hash in Rust: the generator, the acceptance rule, the memory-hard dataset, the CPU verifier and the
kernel emitters. This is the crate the rusty-kaspa fork calls (`docs/fork-map.md`, rows a1 to a3) and, since
4 October 2026, the source of every program pack in `proto-cuda/packs/`. No dependency outside the standard
library; `serde_json` is a dev-dependency for reading the packs in the tests.
Dates: 3 October 2026 (crate, bit-exact with the Swift prototype), 4 October 2026 (generator version 2 and the
acceptance rule; every vector re-cut). Toolchain: rustc 1.99.0 via rustup (the Homebrew 1.69 on PATH is too old;
use `~/.cargo/bin/cargo`). Crate version 0.2.0.
## Modules
| Module | What it is | Swift namesake |
|---|---|---|
| `seed` | 32-byte seed words from bytes (FNV-1a 64, four salts, finalised); `seed_words_from_bytes` is the boundary where the chain feeds the epoch seed; SplitMix64 | `seedWordsBytes`, `SplitMix64` |
| `generator` | version 2: 16 load slots drawn first from instructions 1..63, fresh-source loads, the other 48 ops from the ten non-load weights; attempts `k = 0, 1, ...` of a seed; the program id. The retired version 1 generator stays as `generate_v1` for the census and the lever measurements | `generateProgramV2`, `candidateProgram`, `generateProgramV1` |
| `accept` | the acceptance rule of spec 01 section 1.4.6: two static tests and the 64-unit dynamic test on the seed-keyed closed-form dataset | `acceptProgram` |
| `memhard` | 256 MiB cache (2^16 chains of 64 ChaCha12 blocks), mixer parameters, 8-round item derivation with the 32 lanes interleaved, `MemhardCpu::fetch` | `cpuFillCache`, `MixParams`, `deriveItems`, `MemhardCPU` |
| `verify` | the 32-lane warp interpreter, `DatasetMode::{ClosedForm, MemoryHard}`, `Epoch`, `hash_warp`, `verify_block` | `cpuWarp`, `DatasetSource` |
| `emit` | Metal, CUDA and OpenCL source, program.h, memhard.h, vectors.h, program.json, vectors.json, the header-bound kernels, `export_pack` | `generateMSL`, `generateCUDA`, `generateOpenCL`, `exportPack` |
| `bind` | header binding (spec 01 section 1.6): init words from `"igneum-block/" \|\| H \|\| nonce_hi_le32`, bound hash API on `Epoch`, the 256-bit pow mapping, the interim day seed bytes | `blockInitWords` |
## Generator version 2 (4 October 2026)
Adopted from `docs/analysis/weak-program-census-2026-10-03.md` (ledger M5 and M6). Three parts, all in this crate
and mirrored in `proto-metal/main.swift` so the Metal worker derives the same program from the same seed:
| Part | Rule | Where |
|---|---|---|
| G1, exact load count | 16 `load` instructions per program, a uniform 16-subset of slots 1..63 drawn first by partial Fisher-Yates over the program stream; the other 48 ops from `add 12, xor 10, mul 8, mad 8, shfl 8, rotl 7, sub 6, mulhi 6, rotr 6, or 4` (sum 75). 128 loads per hash, 4,096 items per unit | `generator::candidate_from_words` |
| G2, fresh source | a load's source is drawn from the registers other than `dst` written by an earlier instruction and not read by a load since, so no load repeats an earlier load's address in the hash | same |
| R, acceptance | (a) no load whose source is unwritten since the previous load from it, cyclically; (b) every register has an `add`, `sub`, `xor`, `mad`, `shfl` or `load` write; (c) 64 units at base nonces from `SplitMix64(FNV-1a-64("igneum-accept/" \|\| seed words LE))`, init words = seed words, closed-form dataset `dataset_elem(idx, S[0], S[1])` at 2^28 words: no constant register bit, no lane-constant load site in any unit, fewer than 164 saturated final values, every output bit within 136 of 1,024, distinct addresses above 245,760 over the 2,048 hashes | `accept::check` |
| Attempts | a rejected candidate is replaced by `seed_words_from_bytes(seed \|\| k_le32)` for `k = 1, 2, ...`; 32 consecutive rejections are a consensus fault (probability below 2^-136 at the measured 5 percent rate) | `generator::generate_from_seed_bytes` |
| Program id | `FNV-1a-64("igneum-program/" \|\| 2_le32 \|\| seed words LE \|\| attempt_le32)`, written into program.json and program.h with the generator version and the attempt, so a version 1 pack or another attempt can never pass for the current program | `Program::program_id` |
Measured on this crate (`igneum-pow accept`): `igneum-genesis` attempt 0 accepted, program id `bcc1248b10cc90f2`,
op mix `load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1`, 128.000 distinct addresses per
hash; `igneum-hourly` attempt 0, id `a4c4d00961c855df`; seeds `igneum-census-2026-10-03/22`, `/37` and `/51` have
attempt 0 rejected ((b) r7 without an injecting write; (c) 119.74 distinct addresses; (b) r4) and attempt 1
accepted (ids `22ed0609d079f4cf`, `947705cc4eb1df0a`, `9869afcc028bf9f1`); those three are the conformance vectors
for the attempt rule. The rule costs 1.3 to 3.4 ms per seed on one core. The 20,000-program census under this
generator is in `docs/bench-log.md` (4 October 2026 entry).
## The API the fork calls
```rust
use igneum_pow::{Epoch, DatasetMode};
// Once per epoch and day: derives the accepted program and fills the 256 MiB cache (about 0.2 s on one core).
let epoch = Epoch::memory_hard("igneum-genesis", "2026-10-03");
let h: u64 = epoch.hash(nonce); // one nonce (computes its aligned 32-nonce warp)
let w: [u64; 32] = epoch.hash_warp(base_nonce); // one warp
let ok: bool = epoch.verify_block(nonce, target_u64);
// Miner programs for the epoch: the pack every worker compiles.
let pack = igneum_pow::emit::export_pack(&epoch, "2026-10-03", "igneum node");
pack.write_to(std::path::Path::new("out"))?; // kernel.cu, kernel.cl, program.metal, memhard.h, ...
```
## The header-bound form (what the chain uses)
The pack form above initialises the lane registers from the program's own seed words, so one nonce has one
hash per epoch whatever block is mined. On the chain the init words commit to the block (spec 01 section 1.6,
`src/bind.rs`):
```
H = header hash with the nonce field zeroed, every other field as mined
(rusty-kaspa hash_override_nonce_time(header, 0, header.timestamp))
nonce = 64 bits; lane nonce n = low 32 bits; nonce_hi = high 32 bits
I = seed_words_from_bytes("igneum-block/" || H || nonce_hi_le32) (49 bytes in)
hash = interpret(program, I, n) (section 1.7 of the spec)
pow256 = hash in the top 64 bits, low 192 bits zero (little-endian bytes 24..32)
valid = pow256 <= target256, which is exactly hash <= target256 >> 192
```
`H` keeps the timestamp (Kaspa zeroes it in the pre-PoW hash and absorbs it in cSHAKE afterwards; the lane hash has
no afterwards, so a nonce would otherwise be reusable across timestamps). The interim day seed is
`"igneum-day/" || day_le64` with `day = timestamp_ms / 86,400,000`. The epoch seed bytes are the 32 bytes of the
epoch block hash (devnet); the program is `generate_from_seed_bytes(epoch_seed)`, attempts included.
```rust
use igneum_pow::{bind, Epoch};
let epoch = Epoch::from_seed_bytes(epoch_hash.as_bytes(), &bind::day_bytes(day), "label");
let lane: u64 = epoch.hash_bound(&prehash, nonce); // one 64-bit nonce
let warp: [u64; 32] = epoch.hash_warp_bound(&prehash, nonce); // its aligned 32-nonce group
let init = bind::block_init_words(&prehash, nonce); // what a GPU kernel takes as its argument
let same = epoch.hash_warp_init(&init, nonce as u32 & !31); // == warp
let pow: [u8; 32] = epoch.pow_bound(&prehash, nonce);
let ok = epoch.verify_block_bound(&prehash, nonce, bind::target64_from_le256(&target_le));
```
On the GPU the init words are a kernel argument: `igneum_hash_bound` in `program_bound.metal` takes
`constant uint* initw [[buffer(3)]]`, `kernel_bound.cu` takes `IgneumInitWords iw` by value, `kernel_bound.cl` an
`initw` buffer. All three differ from `igneum_hash` only in the kernel name, the argument and the eight init lines.
Bound vectors (seed `igneum-genesis`, day `2026-10-03`, memory-hard, 2^28 words, generator v2; `igneum-pow hash-bound`):
| H | nonce | hash_bound |
|---|---|---|
| 32 zero bytes | 0 | `746c567b090acf6a` |
| 32 zero bytes | 1 | `45a619f860880c73` |
| 32 zero bytes | 31 | `9aa495e43dedbfe6` |
| 32 zero bytes | 4096 | `2e6ffd7624d3cba2` |
| 32 zero bytes | 4294967296 (1 << 32) | `38a5cea1fb01431a` |
| bytes 00 01 02 .. 1f | 0 | `2a79c5e4797bf6aa` |
| bytes 00 01 02 .. 1f | 4294967301 ((1 << 32) + 5) | `a243e0c61aa1b82e` |
| bytes 00 01 02 .. 1f | 18446744073709551615 (u64::MAX) | `9c7bbfbd064fe1a4` |
Init words for H = 32 zero bytes, nonce 0: `595a8f8a 37647e95 faadade1 cbbcf2a4 54f7cc13 f6851b5e 8c68ca04 7991ea9c`
(unchanged by v2: the binding does not depend on the program). The eight are pinned in `bind::tests::bound_vectors`.
The version 1 bound vectors of 3 October 2026 are retired.
`Epoch` is `Send + Sync`; build one and share it. `DatasetMode::ClosedForm` is the interpreter regression dataset
of the packs `igneum-genesis` and `igneum-hourly` and is not memory-hard.
## CLI
```
cargo build --release
./target/release/igneum-pow bench --seed igneum-genesis [--warps 20] [--closed-form] [--day 2026-10-03]
./target/release/igneum-pow export --seed igneum-genesis --out <dir> [--closed-form]
./target/release/igneum-pow export --epoch-hex <64 hex> --day-hex <hex> --out <dir> the chain's byte seeds
./target/release/igneum-pow hash --seed igneum-genesis --nonce 4103
./target/release/igneum-pow hash-bound --seed igneum-genesis --prehash <64 hex> --nonce <u64>
./target/release/igneum-pow accept --seed igneum-census-2026-10-03/22 every candidate with its verdict
./target/release/igneum-pow show --seed igneum-genesis the accepted program, one line per instruction
```
The packs were regenerated on 4 October 2026 with exactly these commands:
```
igneum-pow export --closed-form --seed igneum-genesis --out ../proto-cuda/packs/igneum-genesis
igneum-pow export --closed-form --seed igneum-hourly --out ../proto-cuda/packs/igneum-hourly
igneum-pow export --seed igneum-genesis --out ../proto-cuda/packs/igneum-genesis-mh
igneum-pow export --epoch-hex edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07 \
--day-hex 69676e65756d2d6461792ffa50000000000000 --out ../proto-cuda/packs/igneum-devnet-v4-epoch0
```
The last is the chain's own derivation for devnet v4 epoch 0: the devnet genesis hash as the epoch seed and
`bind::day_bytes(20730)` (2026-10-04) as the day bytes.
## Tests
`cargo test --release` (39 tests, about 2 s after compile; the dev profile is optimised so the cache fill is quick):
| Check | Pack | Result |
|---|---|---|
| program.json instruction by instruction from `seed_bytes`, generator 2, attempt, program id, op mix, 128 loads, acceptance | all four | match |
| Mixer parameters (key, rot, mul, rc) | igneum-genesis-mh, igneum-devnet-v4-epoch0 | match |
| Cache head, last line, FNV-1a 64 (`48c4f5bf24166b2e` for day 2026-10-03, `448274a57f508cbc` for day bytes 20730) | the two memory-hard packs | match |
| Dataset head (16), `[MASK]`, 64 sampled words | all four | match |
| 96 hash vectors (3 warps x 32 lanes) | all four | 96/96 each |
| kernel.cu, kernel_bound.cu, program.metal, program_bound.metal, kernel.cl, kernel_bound.cl, program.h, program.json byte-identical; 16 masked loads per kernel | all four | identical |
| memhard.h, memhard.metal byte-identical | the two memory-hard packs | identical |
| vectors.json, vectors.h byte-identical; no stale file in any pack directory | all four | identical |
| bound vectors (8), bound warp == bound single, H and nonce_hi enter the hash | igneum-genesis-mh | pass |
| the devnet pack equals `Epoch::from_seed_bytes(genesis, day_bytes(20730))` | igneum-devnet-v4-epoch0 | pass |
| acceptance: instrumented interpreter == `hash_warp`, cyclic stale-load detection, injecting-write detection, rejection under 12.5 percent and distinct loads above 127 on 400 census seeds, version 1 programs mostly rejected | unit tests | pass |
| generator: 16 loads, none at instruction 0, contract on 200 candidates; fresh sources; attempt words; program ids separate versions and attempts | unit tests | pass |
## Measured, Apple M5 Max, one core, release build
| Step | 3 October 2026 (v1, 104 loads) | 4 October 2026 (v2, 128 loads, 4,096 items) |
|---|---|---|
| Cache fill, 256 MiB | 175 to 181 ms | 179 ms |
| CPU verify per warp, igneum-genesis, avg of 20 | 0.441 ms (3,328 items) | 0.631 ms |
| Cold single warps (bases 0, 4096, 1000000) | 0.41 to 0.87 ms | 0.67 to 0.81 ms |
| Acceptance rule per candidate | | 1.3 to 3.4 ms |
| Closed form, igneum-genesis | 0.002 ms | 0.002 ms |
The 10 ms gate holds with a margin of about 16x steady on this core. Every verified unit now derives exactly 4,096
items, the design bound of spec section 1.11.
## Not done here
- No GPU. The vectors tie this crate to the Metal, CUDA and OpenCL results through the packs; nothing here runs a kernel.
- The epoch seed enters at `Epoch::from_seed_bytes` (the epoch block hash on devnet); the VDF output replaces it there.
- The dynamic acceptance test is specified on the closed-form dataset at 2^28 words; if the prototype mask ever changes, the rule's constant stays at 2^28.

View file

@ -0,0 +1 @@
class-v5 igneum-pow at commit 25c8063ff38e248b7a92d91575e975a81d953894

View file

@ -0,0 +1,2 @@
max_width = 120
use_small_heuristics = "Max"

View file

@ -0,0 +1,869 @@
//! Program acceptance (spec 01 section 1.4.6, adopted 4 October 2026 from the weak-program census of
//! `docs/analysis/weak-program-census-2026-10-03.md`, section 6).
//!
//! A candidate program is accepted only if every test below holds. Every conforming implementation evaluates
//! exactly these tests on exactly these inputs, so every node skips the same seeds.
//!
//! | Part | Test |
//! |---|---|
//! | (a) static | for every `load`, some instruction between the previous `load` from the same source register and this one, in cyclic order over the 64 instructions, writes that register |
//! | (b) static | every register `r0..r7` is the destination of at least one `add`, `sub`, `xor`, `mad`, `shfl` or `load` |
//! | (c) dynamic | the program is interpreted for [`ACCEPT_UNITS`] (64) units of 32 lanes at base nonces drawn from SplitMix64 seeded with `FNV-1a-64("igneum-accept/" \|\| seed words as little-endian bytes)`, each `low32(next()) AND NOT 31`, with init words equal to the seed words and the closed-form dataset `dataset_elem(idx, S[0], S[1])` at [`ACCEPT_DATASET_LOG2`] (2^28 words) in place of the memory-hard dataset. Over the 2,048 evaluations: no register has a bit equal in every final value; no load site (iteration, instruction) reads one address in all 32 lanes of any unit; fewer than [`MAX_SATURATED`] (164, 1 percent of 16,384) final register values are 0 or 2^32 - 1; every output bit's ones count is within [`BIAS_TOLERANCE`] (136, 6 sigma) of 1,024; the distinct masked addresses read by one lane in one evaluation, summed over the 2,048 evaluations, exceed [`MIN_DISTINCT_SUM`] (245,760, a mean above 120 of the 128 loads) |
//!
//! The dynamic test uses the closed form so that it is a pure function of the program (no cache, no day) and
//! costs about a millisecond on one core. A hot-table load (`docs/plans/hot-table.md`) reads the closed form keyed by
//! seed words 2 and 3 at its multiply-shift index, a second pure table beside the dataset stand-in (words 0 and 1). The census (section 7.3) checked on 100,000 programs that the
//! closed-form verdict agrees with the memory-hard one on all but 39 threshold-edge cases.
use crate::generator::{Instr, LoadClass, Op, Program, ShadowClass, INSTR_COUNT, ITERATIONS, LANES, V4_CLASS, V4_SHADOW_INSTRS};
use crate::seed::{fnv1a64, SplitMix64};
use crate::memhard::hot_index;
use crate::verify::{dataset_elem, fold_words, load_index, splitmix32, ScratchModel};
/// Units (32-lane warps) the dynamic test interprets.
pub const ACCEPT_UNITS: usize = 64;
/// Class v4 sub-version 3, rule (c''): the per-site distinct-index RATIO (AP-F8-1's low-entropy-band class, 7 October
/// 2026, the attack-pass gate's numbers through main), keyed on the class v4 shape. Over [`ACCEPT_UNITS_DISTINCT_V4`]
/// units (2^20 evaluations per site) the count of distinct dataset word indices a load site reads, against the
/// expectation of a uniform source on the site's window (N - N^2 / 2W), must reach [`MIN_DISTINCT_RATIO_V4`]. The
/// floor sits between the 55 clean F8 seeds' minimum over their site rows (0.9960; the clean p1 0.9990, the median
/// 1.0000) and the strong failing seeds' maximum (p56 0.9654; p23 0.8361, p18 0.9274, p19 0.9335, p15 0.9432),
/// 0.015 from each. The open tail: F8's p4, p8, p10 and p34 (1.22x to 1.50x on the gate) read 0.9927 to 0.9963 at
/// 2^20, inside the clean spread, and a 2^24 pass does not separate them either (p34 0.9181, p4 0.9614, p8 0.9630,
/// p10 0.9612 against the clean p44 0.9612, p52 0.9613, p3 0.9971, p2 and p5 1.0004); they stay unattributed and
/// chased in `docs/fud-ledger.md` AP-F8-1. A candidate under the floor is rejected and the next attempt drawn under
/// the 256 cap and the last resort. Measured on one box-2 core with the shadow executed: 2.8 s per chosen candidate.
pub const ACCEPT_UNITS_DISTINCT_V4: usize = 4096;
/// The ratio floor at 2^20.
pub const MIN_DISTINCT_RATIO_V4: f64 = 0.98;
/// Kept for the record and the driver, not wired: the most repeated source value per site over the (c) units'
/// 16,384 evaluations (a uniform site repeats a value 2 or 3 times; the finding's bands sit under the ratio instead).
pub const MAX_SOURCE_REPEAT_V4: u32 = 8;
/// Hashes the dynamic test evaluates: 2,048.
pub const ACCEPT_HASHES: usize = ACCEPT_UNITS * LANES;
/// Domain tag of the base-nonce stream.
pub const ACCEPT_TAG: &[u8] = b"igneum-accept/";
/// log2 of the closed-form dataset the test addresses: the prototype's 2^28 words, MASK 0x0fffffff.
pub const ACCEPT_DATASET_LOG2: u32 = 28;
/// Final register values equal to 0 or 2^32 - 1 must number fewer than this (1 percent of 8 x 2,048).
pub const MAX_SATURATED: u32 = 164;
/// Every output bit's ones count must be within this of 1,024 (6 x sqrt(2048) / 2, rounded).
pub const BIAS_TOLERANCE: u32 = 136;
/// Distinct addresses per lane per evaluation, summed over 2,048 evaluations, must exceed this (mean above 120).
pub const MIN_DISTINCT_SUM: u64 = 245_760;
/// The distinct-address bound for a program with `loads` dataset loads per hash: the same 120 of 128 ratio, so
/// [`MIN_DISTINCT_SUM`] for the lottery hash and `loads x 1,920` for the read-width classes with other counts.
/// Variant 5's scratch read-modify-writes are not dataset loads: their slots repeat by design (a later
/// read-modify-write sees an earlier write), so they are neither counted nor bounded here.
pub fn min_distinct_sum(loads: usize) -> u64 {
loads as u64 * ACCEPT_HASHES as u64 * 120 / 128
}
/// Why a candidate was rejected. The verdict (accept or reject) is what consensus depends on; the reason is the
/// first failing test in the order of the module table.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum Reject {
/// (a): instruction `instr` loads from `reg`, which no instruction wrote since the previous load from it.
StaleLoadSource { instr: u8, reg: u8 },
/// (b): no injecting op writes `reg`.
NoInjectingWrite { reg: u8 },
/// (c): `reg` has `bits` bits equal in all 2,048 final values.
ConstantBit { reg: u8, bits: u8 },
/// (c): the load at `instr` in `iteration` read one address in all 32 lanes of `unit`.
LaneConstantSite { iteration: u8, instr: u8, unit: u8 },
/// (c): `count` final register values were 0 or all ones.
Saturated { count: u32 },
/// (a'), class v4 sub-version 2 (AP-F8-1): the load at `instr` reads `reg`, which is not fresh by dataflow in the
/// steady state of the loop (the freshness fixpoint over the base program and the shadow block).
UnfreshLoadSource { instr: u8, reg: u8 },
/// (c'), class v4 sub-version 2 (AP-F8-1): the load at `site` read a source value of 0 or all ones in `count` of
/// its 16,384 evaluations (64 units x 32 lanes x 8 iterations); limit [`MAX_SATURATED`] - 1, the same 1 percent as (c).
SaturatedSource { site: u8, count: u32 },
/// (c'') (B), class v4 sub-version 3: the load at `site` read the value `value` in `count` of its 16,384 (c)
/// evaluations (limit [`MAX_SOURCE_REPEAT_V4`] - 1): one constant upstream that the lineage rule cannot see.
RepeatedSource { site: u8, value: u32, count: u32 },
/// (c''), class v4 sub-version 3: the load at `site` read `distinct` distinct dataset word indices over
/// `evaluations`, `ratio_milli` / 1000 of a uniform source on its window, under the floor: a low-entropy index band
/// (F8's p23, p18, p19, p15, p56).
LowEntropySite { site: u8, distinct: u32, evaluations: u32, ratio_milli: u32 },
/// (c): output bit `bit` was set in `ones` of 2,048 hashes.
OutputBias { bit: u8, ones: u32 },
/// (c): the distinct-address sum was `sum`.
DistinctAddresses { sum: u64 },
}
impl std::fmt::Display for Reject {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
Reject::StaleLoadSource { instr, reg } => {
write!(f, "(a) load at instruction {instr} reads r{reg}, unwritten since the previous load from it")
}
Reject::NoInjectingWrite { reg } => write!(f, "(b) r{reg} has no add, sub, xor, mad, shfl or load write"),
Reject::ConstantBit { reg, bits } => write!(f, "(c) r{reg} has {bits} nonce-independent bits"),
Reject::LaneConstantSite { iteration, instr, unit } => {
write!(f, "(c) load at iteration {iteration} instruction {instr} reads one address in all lanes of unit {unit}")
}
Reject::Saturated { count } => write!(f, "(c) {count} of 16384 final register values saturated (limit 163)"),
Reject::UnfreshLoadSource { instr, reg } => write!(f, "(a') load at {instr} reads r{reg}, not fresh by dataflow in the loop's steady state (class v4 sub-version 2)"),
Reject::RepeatedSource { site, value, count } => write!(f, "(c'') load site {site} read the value {value:#010x} in {count} of 16384 evaluations (limit {})", MAX_SOURCE_REPEAT_V4 - 1),
Reject::LowEntropySite { site, distinct, evaluations, ratio_milli } => write!(f, "(c'') load site {site} read {distinct} distinct word indices over {evaluations} evaluations, {}.{:03} of a uniform source on its window (floor {MIN_DISTINCT_RATIO_V4} at 2^20)", ratio_milli / 1000, ratio_milli % 1000),
Reject::SaturatedSource { site, count } => write!(f, "(c') load site {site} read a saturated source value in {count} of 16384 evaluations (limit 163)"),
Reject::OutputBias { bit, ones } => write!(f, "(c) output bit {bit} set in {ones} of 2048 hashes"),
Reject::DistinctAddresses { sum } => {
write!(f, "(c) distinct dataset addresses {sum} over 2048 hashes (mean {:.2}, needs above 120 of 128 of the dataset loads)", *sum as f64 / 2048.0)
}
}
}
}
/// What the dynamic test measured on an accepted program.
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
pub struct AcceptReport {
/// Distinct masked addresses per lane per evaluation, summed over the 2,048 evaluations.
pub distinct_sum: u64,
/// Final register values equal to 0 or all ones.
pub saturated: u32,
/// The largest `|ones - 1024|` over the 64 output bits.
pub bias_max: u32,
}
impl AcceptReport {
/// Mean distinct addresses per hash (128 at most).
pub fn distinct_mean(&self) -> f64 {
self.distinct_sum as f64 / ACCEPT_HASHES as f64
}
}
/// Part (a): no load whose source is unwritten since the previous load from it, cyclically.
fn check_stale_loads(instrs: &[Instr]) -> Result<(), Reject> {
// `pending[r]`: a load has read r and nothing has written r since. Two passes over the list so the second
// pass sees the state carried over the iteration boundary.
let mut pending = [false; 8];
for _pass in 0..2 {
for (k, ins) in instrs.iter().enumerate() {
if ins.op.is_load() && pending[ins.src as usize] {
return Err(Reject::StaleLoadSource { instr: k as u8, reg: ins.src });
}
pending[ins.dst as usize] = false;
if ins.op.is_load() {
pending[ins.src as usize] = true;
}
}
}
Ok(())
}
/// Part (b): every register has an injecting write.
fn check_injecting_writes(instrs: &[Instr]) -> Result<(), Reject> {
let mut injected = [false; 8];
for ins in instrs {
if ins.op.injects() {
injected[ins.dst as usize] = true;
}
}
for (reg, ok) in injected.iter().enumerate() {
if !ok {
return Err(Reject::NoInjectingWrite { reg: reg as u8 });
}
}
Ok(())
}
/// The most repeated value of `values` (sorted in place) and that value: (B), kept for the driver, not wired.
#[allow(dead_code)]
fn most_repeated(values: &mut [u32]) -> (u32, u32) {
values.sort_unstable();
let (mut best, mut best_v, mut run) = (0u32, 0u32, 0u32);
for i in 0..values.len() {
run = if i > 0 && values[i] == values[i - 1] { run + 1 } else { 1 };
if run > best {
best = run;
best_v = values[i];
}
}
(best, best_v)
}
/// (c''), class v4 sub-version 3: the 2^20 ratio pass on the chosen candidate (the constants above).
pub fn check_distinct_indices_v4(p: &Program) -> Result<(), Reject> {
distinct_ratio_pass(p, ACCEPT_UNITS_DISTINCT_V4, MIN_DISTINCT_RATIO_V4).map(|_| ())
}
/// One ratio pass over `units`: every load site's distinct word indices against the uniform expectation on its
/// window (`N - N^2 / 2W`, the window `2^28 >> min(win, 2)` words of the closed-form dataset), `Err` at the first
/// site under `floor`, else the minimum ratio and its site.
pub fn distinct_ratio_pass(p: &Program, units: usize, floor: f64) -> Result<(f64, usize), Reject> {
let n = (units * LANES * ITERATIONS) as f64;
let d = distinct_indices_v4(p, units)?;
let mut min = (f64::MAX, 0usize);
let mut site = 0usize;
for i in &p.instrs {
if !i.op.is_load() {
continue;
}
let wsize = ((1u64 << ACCEPT_DATASET_LOG2) >> (i.win as u64).min(2)) as f64;
let ratio = d[site] as f64 / (n - n * n / (2.0 * wsize));
if ratio < floor {
return Err(Reject::LowEntropySite { site: site as u8, distinct: d[site], evaluations: n as u32, ratio_milli: (ratio * 1000.0) as u32 });
}
if ratio < min.0 {
min = (ratio, site);
}
site += 1;
}
Ok(min)
}
/// The distinct dataset word indices every load site reads over `units` units of the seed's acceptance stream on
/// the closed-form words (the sample caps the count near `units x 32 x 8`, so a site's index entropy is read only
/// below about log2 of that).
pub fn distinct_indices_v4(p: &Program, units: usize) -> Result<Vec<u32>, Reject> {
let loads = p.loads_per_hash();
let sites = loads / ITERATIONS;
let mut acc = Acc {
sources: None,
indices: Some(vec![Vec::with_capacity(units * LANES * ITERATIONS); sites]),
sat_source: vec![0; sites],
and_acc: [u32::MAX; 8],
or_acc: [0; 8],
saturated: 0,
bit_ones: [0; 64],
distinct_sum: 0,
};
let mut lane_addrs = vec![0u32; LANES * loads];
for (unit, &base) in accept_base_nonces_n(&p.seed, units).iter().enumerate() {
run_unit(p, unit, base, &mut acc, &mut lane_addrs)?;
}
let mut out = Vec::with_capacity(sites);
for ix in acc.indices.take().unwrap().iter_mut() {
ix.sort_unstable();
ix.dedup();
out.push(ix.len() as u32);
}
Ok(out)
}
/// Whether `class` is the class v4 shape (the 256-instruction shadow block over the class v3 base, the pass count and
/// the era set aside): the shape the sub-version 2 rules (a') and (c') apply to, on every draw path.
pub fn is_class_v4_shape(class: &LoadClass) -> bool {
matches!(class.shadow, Some(ShadowClass { instrs: V4_SHADOW_INSTRS, .. }))
// class v5 (docs/design/class-v5-stored-state.md) is judged under the same rules: its state flag is set aside
&& LoadClass { era: None, shadow: None, state: false, ..*class } == LoadClass { shadow: None, ..V4_CLASS }
}
/// One pass of the dataflow freshness over the base program then the shadow block (the order of one iteration),
/// from `fresh`; `check` reports the first load that reads a register that is not fresh. The rule (AP-F8-1,
/// `docs/analysis/ca3-v4-uniform.md`): a load leaves its destination fresh only if its source was (a saturated
/// source reads one fixed word); add, sub, xor, mad and shfl if either operand was; rotl and rotr if the operand
/// was (a rotate maps all-ones and zero to themselves); or, mul and mulhi never.
fn freshness_pass(p: &Program, fresh: &mut [bool; 8], pair_op: &mut [Option<(Op, usize)>; 8], check: bool) -> Result<(), Reject> {
for (k, i) in p.instrs.iter().chain(p.shadow.iter()).enumerate() {
let (d, a) = (i.dst as usize, i.src as usize);
if check && i.op.is_load() && !fresh[a] {
return Err(Reject::UnfreshLoadSource { instr: k as u8, reg: i.src });
}
// the shared-operand idiom (sub-version 3): or-then-xor or or-then-sub on one operand is `d & ~s`, xor-then-or
// is `d | s`: lossy, though the second op would inject on its own (F8's p23: `or r6 |= r4; xor r6 ^= r4`)
let masked = matches!((pair_op[d], i.op), (Some((Op::Or, s)), Op::Xor) | (Some((Op::Or, s)), Op::Sub) | (Some((Op::Xor, s)), Op::Or) if s == a);
fresh[d] = !masked
&& match i.op {
Op::Load | Op::WLoad | Op::Scratch | Op::Hot => fresh[a],
Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl => fresh[d] || fresh[a],
Op::Rotl | Op::Rotr => fresh[d],
Op::Or | Op::Mul | Op::MulHi => false,
};
pair_op[d] = if matches!(i.op, Op::Or | Op::Xor) && !masked { Some((i.op, a)) } else { None };
for r in 0..8 {
if r != d {
if let Some((_, s)) = pair_op[r] {
if s == d {
pair_op[r] = None;
}
}
}
}
}
Ok(())
}
/// Part (a'), class v4 sub-version 2: every load's source is fresh by dataflow in the loop's steady state. The draw
/// of `candidate_from_words_class` keeps in-pass sources fresh; this closes the iteration boundary (a source last
/// written late in the previous iteration or in the shadow block, which the draw's no-eligible fallback can pick:
/// F8's p11, an `or` at 63 feeding a load at 1). The state starts all fresh (the init words are a per-lane hash of
/// the nonce) and is run to its fixpoint (it only ever falls, so at most 8 passes change it), then one checking pass.
pub fn check_fresh_sources_v4(p: &Program) -> Result<(), Reject> {
if !is_class_v4_shape(&p.class) {
return Ok(());
}
let mut fresh = [true; 8];
let mut pair_op: [Option<(Op, usize)>; 8] = [None; 8];
for _ in 0..9 {
let before = (fresh, pair_op);
freshness_pass(p, &mut fresh, &mut pair_op, false)?;
if (fresh, pair_op) == before {
break;
}
}
freshness_pass(p, &mut fresh, &mut pair_op, true)
}
/// Parts (a), (b) and, for class v4 sub-version 2, (a').
pub fn check_static(p: &Program) -> Result<(), Reject> {
if p.instrs.len() != INSTR_COUNT {
panic!("acceptance needs a {INSTR_COUNT}-instruction program");
}
check_stale_loads(&p.instrs)?;
check_injecting_writes(&p.instrs)?;
check_fresh_sources_v4(p)
}
/// The 64 base nonces of the dynamic test for seed words `seed`.
pub fn accept_base_nonces(seed: &[u32; 8]) -> [u32; ACCEPT_UNITS] {
let v = accept_base_nonces_n(seed, ACCEPT_UNITS);
let mut out = [0u32; ACCEPT_UNITS];
out.copy_from_slice(&v);
out
}
/// The first `n` base nonces of the seed's acceptance stream (the (c) units are the first [`ACCEPT_UNITS`]).
pub fn accept_base_nonces_n(seed: &[u32; 8], n: usize) -> Vec<u32> {
let mut b = Vec::with_capacity(ACCEPT_TAG.len() + 32);
b.extend_from_slice(ACCEPT_TAG);
for w in seed {
b.extend_from_slice(&w.to_le_bytes());
}
let mut rng = SplitMix64::new(fnv1a64(&b));
(0..n).map(|_| (rng.next() as u32) & !31).collect()
}
#[inline(always)]
fn mulhi32(a: u32, b: u32) -> u32 {
((a as u64 * b as u64) >> 32) as u32
}
/// Accumulators of the dynamic test over the 64 units.
struct Acc {
/// (c'') (B): every load's source value per site, recorded when present.
sources: Option<Vec<Vec<u32>>>,
/// (c'') (A): every load's dataset index per site, recorded when present (the distinct-index pass only).
indices: Option<Vec<Vec<u32>>>,
/// (c'): per load site (the load's index within the iteration), how many of its evaluations read a source value
/// of 0 or all ones (class v4 sub-version 2; counted for every class, judged for class v4 only).
sat_source: Vec<u32>,
and_acc: [u32; 8],
or_acc: [u32; 8],
saturated: u32,
bit_ones: [u32; 64],
distinct_sum: u64,
}
/// One unit of the dynamic test: the interpreter of `verify.rs` with the closed-form dataset, instrumented.
/// Returns the first lane-constant load site, if any.
fn run_unit(p: &Program, unit: usize, base: u32, acc: &mut Acc, lane_addrs: &mut [u32]) -> Result<(), Reject> {
let seed = &p.seed;
let mask: u32 = (1u32 << ACCEPT_DATASET_LOG2) - 1;
let (d0, d1) = (seed[0], seed[1]);
let (h0, h1) = (seed[2], seed[3]);
let hot_words = p.hot_words();
let loads = p.loads_per_hash();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
let nonce = base.wrapping_add(lane as u32);
for i in 0..8 {
let mut x = nonce ^ seed[i];
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
x = splitmix32(x);
r[i][lane] = x ^ seed[(i + 1) & 7];
}
}
let mut idx = [0u32; LANES];
let mut nload = 0usize;
let mut scratch = if p.has_scratch() { Some(ScratchModel::new(p.class.scratch_slots_per_lane())) } else { None };
let slot_mask = p.class.scratch_slot_mask();
let era = p.class.era;
// Class v4 sub-version 3 (AP-F8-3, 7 October 2026): the acceptance interpreter runs the latency-shadow block
// after instruction 63 of every iteration, `reps` times with the iteration's `sel`, exactly as the hash does
// (verify.rs). Until this commit it ran the 64 base instructions only, so every dynamic test (c) judged a class v4
// program the chain never hashes. The shadow block holds no load, so its instructions take the same arms.
let shadow_reps = p.shadow_reps();
for it in 0..ITERATIONS {
let sel = r[0];
let shadow_pass = (0..shadow_reps).flat_map(|_| p.shadow.iter().enumerate().map(|(k, i)| (INSTR_COUNT + k, i)));
for (k, ins) in p.instrs.iter().enumerate().chain(shadow_pass) {
let d = ins.dst as usize;
let a = ins.src as usize;
match ins.op {
Op::Scratch => {
// Variant 5: the slot stands in for the address (bit 31 set so it never aliases a dataset word).
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
for lane in 0..LANES {
idx[lane] = r[a][lane] & slot_mask;
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
r[d][lane] = m.rmw(&p.seed, base, lane, idx[lane], r[d][lane]);
lane_addrs[lane * loads + nload] = 0x8000_0000 | idx[lane];
}
nload += 1;
}
Op::Add => {
let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32);
let src = r[a];
for lane in 0..LANES {
let s = (sel[lane] >> bit) & 1;
let c = if s != 0 { imm2 } else { imm };
r[d][lane] = r[d][lane].wrapping_add(src[lane]).wrapping_add(c);
}
}
Op::Sub => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_sub(src[lane]);
}
}
Op::Mul => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_mul(src[lane]);
}
}
Op::MulHi => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = mulhi32(r[d][lane], src[lane]);
}
}
Op::Xor => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] ^= src[lane];
}
}
Op::Or => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] |= src[lane];
}
}
Op::Rotl => {
let n = ins.rot;
for lane in 0..LANES {
r[d][lane] = r[d][lane].rotate_left(n);
}
}
Op::Rotr => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].rotate_right(src[lane] & 31);
}
}
Op::Mad => {
let src = r[a];
let src2 = r[ins.src2 as usize];
for lane in 0..LANES {
r[d][lane] = src[lane].wrapping_mul(src2[lane]).wrapping_add(r[d][lane]);
}
}
Op::Shfl => {
let src = r[a];
let m = ins.mask as usize;
for lane in 0..LANES {
r[d][lane] ^= src[lane ^ m];
}
}
Op::Load => {
// Read-width experiment: a load of `width` words reads from the aligned address and folds every
// word (verify::fold_words); width 1 is the lottery hash's xor of one word.
let width = ins.width as usize;
let align = !(ins.width as u32 - 1);
let site = nload % (loads / ITERATIONS);
for lane in 0..LANES {
let v = r[a][lane];
acc.sat_source[site] += (v == 0 || v == u32::MAX) as u32;
if let Some(src) = acc.sources.as_mut() {
src[site].push(v);
}
idx[lane] = load_index(era.as_ref(), ins, v, mask, ACCEPT_DATASET_LOG2) & align;
if let Some(ix) = acc.indices.as_mut() {
ix[site].push(idx[lane]);
}
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
if width == 1 {
r[d][lane] ^= dataset_elem(idx[lane], d0, d1);
} else {
let mut w = [0u32; 16];
for j in 0..width {
w[j] = dataset_elem(idx[lane] + j as u32, d0, d1);
}
r[d][lane] = fold_words(r[d][lane], &w[..width]);
}
lane_addrs[lane * loads + nload] = idx[lane];
}
nload += 1;
}
Op::Hot => {
// Hot table: the stand-in is dataset_elem keyed by seed words 2 and 3; the address is tagged with
// bit 30 so a hot word and a dataset word at one index count as two addresses.
for lane in 0..LANES {
idx[lane] = hot_index(r[a][lane], hot_words);
}
if idx.iter().all(|&x| x == idx[0]) {
return Err(Reject::LaneConstantSite { iteration: it as u8, instr: k as u8, unit: unit as u8 });
}
for lane in 0..LANES {
r[d][lane] ^= dataset_elem(idx[lane], h0, h1);
lane_addrs[lane * loads + nload] = 0x4000_0000 | idx[lane];
}
nload += 1;
}
Op::WLoad => {
let b = (r[a][0] & mask) & !31;
for lane in 0..LANES {
idx[lane] = b + lane as u32;
r[d][lane] ^= dataset_elem(idx[lane], d0, d1);
lane_addrs[lane * loads + nload] = idx[lane];
}
nload += 1;
}
}
}
}
for i in 0..8 {
for lane in 0..LANES {
let v = r[i][lane];
acc.and_acc[i] &= v;
acc.or_acc[i] |= v;
acc.saturated += (v == 0 || v == u32::MAX) as u32;
}
}
for lane in 0..LANES {
let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21);
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
let h = ((hi as u64) << 32) | lo as u64;
for j in 0..64 {
acc.bit_ones[j] += ((h >> j) & 1) as u32;
}
let sl = &mut lane_addrs[lane * loads..(lane + 1) * loads];
sl.sort_unstable();
let mut distinct = 0u64;
for k in 0..loads {
// scratch slots carry bit 31 (variant 5) and are not dataset addresses
if sl[k] & 0x8000_0000 == 0 && (k == 0 || sl[k] != sl[k - 1]) {
distinct += 1;
}
}
acc.distinct_sum += distinct;
}
Ok(())
}
/// Part (c).
pub fn check_dynamic(p: &Program) -> Result<AcceptReport, Reject> {
let loads = p.loads_per_hash();
let v4 = is_class_v4_shape(&p.class);
let sites = loads / ITERATIONS;
let mut acc = Acc {
sources: None,
indices: None,
sat_source: vec![0; sites],
and_acc: [u32::MAX; 8],
or_acc: [0; 8],
saturated: 0,
bit_ones: [0; 64],
distinct_sum: 0,
};
let mut lane_addrs = vec![0u32; LANES * loads];
for (unit, &base) in accept_base_nonces(&p.seed).iter().enumerate() {
run_unit(p, unit, base, &mut acc, &mut lane_addrs)?;
}
for reg in 0..8 {
let bits = (acc.and_acc[reg] | !acc.or_acc[reg]).count_ones();
if bits != 0 {
return Err(Reject::ConstantBit { reg: reg as u8, bits: bits as u8 });
}
}
if acc.saturated >= MAX_SATURATED {
return Err(Reject::Saturated { count: acc.saturated });
}
// (c'), class v4 sub-version 2 (AP-F8-1, 7 October 2026): a load whose source is saturated reads one fixed word,
// whatever delivered the saturation (an or-written value, a rotate of one, a load after a saturated load); the
// source rule of the draw removes the writers it can see and this count catches every delivery. Keyed on the
// class v4 shape as the draw's rule is, so v2 and v3 verdicts do not move.
if v4 {
if let Some((site, &count)) = acc.sat_source.iter().enumerate().find(|(_, &c)| c >= MAX_SATURATED) {
return Err(Reject::SaturatedSource { site: site as u8, count });
}
// (c''), the ratio on the candidate that passed everything else (the draw's last and dearest test)
check_distinct_indices_v4(p)?;
}
let half = (ACCEPT_HASHES / 2) as u32;
let mut bias_max = 0u32;
for (bit, &ones) in acc.bit_ones.iter().enumerate() {
let d = ones.abs_diff(half);
if d > BIAS_TOLERANCE {
return Err(Reject::OutputBias { bit: bit as u8, ones });
}
bias_max = bias_max.max(d);
}
if acc.distinct_sum <= min_distinct_sum(loads - p.scratch_ops_per_hash()) {
return Err(Reject::DistinctAddresses { sum: acc.distinct_sum });
}
Ok(AcceptReport { distinct_sum: acc.distinct_sum, saturated: acc.saturated, bias_max })
}
/// The whole rule: (a), (b), then (c).
pub fn check(p: &Program) -> Result<AcceptReport, Reject> {
check_static(p)?;
check_dynamic(p)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::generator::{candidate, candidate_class, generate, generate_class, GeneratorConfig, generate_v1, LoadClass};
use crate::verify::{DatasetMode, DatasetSource};
#[test]
fn distinct_bound_scales_with_the_load_count() {
assert_eq!(min_distinct_sum(128), MIN_DISTINCT_SUM);
assert_eq!(min_distinct_sum(32), 61_440);
}
/// The read-width classes pass the rule at about the version 2 rate, and the instrumented interpreter agrees
/// with `verify.rs` on every class (the fold is shared, the addresses are aligned the same way).
#[test]
fn classes_pass_and_match_verify() {
for name in ["w16", "w64", "w64x4", "50,35,15", "25,50,25", "scr2k32", "scr8k128"] {
let c = LoadClass::parse(name).unwrap();
let p = generate_class("igneum-genesis", c);
assert!(check(&p).is_ok(), "{name}");
let mut rejected = 0;
for i in 0..60u32 {
let s = format!("igneum-rw-accept/{i}");
let q = candidate_class(&s, s.as_bytes(), 0, c);
if check(&q).is_err() {
rejected += 1;
}
}
assert!(rejected < 15, "{name}: {rejected} of 60 rejected");
let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2);
let bases = accept_base_nonces(&p.seed);
let loads = p.loads_per_hash();
let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
let mut la = vec![0u32; LANES * loads];
let mut ones = [0u32; 64];
for (u, &b) in bases.iter().enumerate() {
run_unit(&p, u, b, &mut acc, &mut la).unwrap();
for h in crate::verify::hash_warp(&p, b, &ds) {
for j in 0..64 {
ones[j] += ((h >> j) & 1) as u32;
}
}
}
assert_eq!(acc.bit_ones, ones, "{name}: bit counts match the reference interpreter");
}
}
/// AP-F8-3 (7 October 2026): the acceptance's execution and `verify.rs` agree on a class v4 program WITH its
/// shadow block (the output bit counts over the 64 units on the closed-form dataset, the same sel per iteration),
/// so the two paths cannot diverge again: until sub-version 3 the acceptance ran the base program only and judged
/// a program the chain never hashes. The devnet epoch-0 seed and the six test eras, 8 x 256 x 27 shadow
/// instructions per hash each; the same program with its shadow stripped gives other counts.
#[test]
fn acceptance_executes_the_shadow_block_as_the_verifier_does() {
use crate::generator::{generate_era, EraParams, V3_ALLOWED, V4_CLASS};
let hx = |h: &str| -> Vec<u8> { (0..h.len()).step_by(2).map(|i| u8::from_str_radix(&h[i..i + 2], 16).unwrap()).collect() };
let g = hx("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07");
let mut eras = vec![g.clone()];
for n in 0..6 {
eras.push(EraParams::test_era_bytes(&format!("igneum-era-test/{n}")).to_vec());
}
for era in &eras {
let p = generate_era("igneum-epoch/edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07", &g, V4_CLASS, era, &V3_ALLOWED);
assert_eq!(p.shadow.len(), 256);
assert_eq!(p.shadow_reps(), 27);
let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2);
let bases = accept_base_nonces(&p.seed);
let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
let mut la = vec![0u32; LANES * p.loads_per_hash()];
let mut ones = [0u32; 64];
for (u, &b) in bases.iter().enumerate() {
run_unit(&p, u, b, &mut acc, &mut la).unwrap();
for h in crate::verify::hash_warp(&p, b, &ds) {
for j in 0..64 {
ones[j] += ((h >> j) & 1) as u32;
}
}
}
assert_eq!(acc.bit_ones, ones, "the acceptance's execution of a class v4 program (shadow block included) matches the verifier's hashes");
// and the same program with its shadow stripped hashes differently: the shadow is executed, not skipped
let mut bare = p.clone();
bare.shadow.clear();
let mut acc2 = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
for (u, &b) in bases.iter().enumerate() {
let _ = run_unit(&bare, u, b, &mut acc2, &mut la);
}
assert_ne!(acc.bit_ones, acc2.bit_ones, "the shadow block changes the acceptance's execution");
}
}
/// The instrumented interpreter agrees with `verify.rs` on the closed-form dataset keyed by the seed words.
#[test]
fn instrumented_interpreter_matches_verify() {
for i in 0..20u32 {
let s = format!("igneum-accept-test/{i}");
let p = candidate(&s, s.as_bytes(), 0);
let ds = DatasetSource::from_key(p.seed, DatasetMode::ClosedForm, ACCEPT_DATASET_LOG2);
let bases = accept_base_nonces(&p.seed);
let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
let mut la = vec![0u32; LANES * p.loads_per_hash()];
let mut ones = [0u32; 64];
let mut any = false;
for (u, &b) in bases.iter().enumerate() {
if run_unit(&p, u, b, &mut acc, &mut la).is_err() {
continue;
}
any = true;
let w = crate::verify::hash_warp(&p, b, &ds);
for h in w {
for j in 0..64 {
ones[j] += ((h >> j) & 1) as u32;
}
}
}
if any {
assert_eq!(acc.bit_ones, ones, "bit counts of {s} match the reference interpreter's hashes");
}
}
}
/// Hot-table experiment: the hot classes pass the rule at about the version 2 rate, and the hot addresses are
/// uniform over the table (16 buckets of the index over 64 units x 32 lanes x 32 hot loads).
#[test]
fn hot_classes_pass_and_hot_loads_are_uniform() {
for name in ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "scr4k32+hot64k4", "hot32k4a", "hot64k4a", "hot96k4a"] {
let c = LoadClass::parse(name).unwrap();
let p = generate_class("igneum-genesis", c);
assert!(check(&p).is_ok(), "{name}");
let mut rejected = 0;
for i in 0..60u32 {
let s = format!("igneum-hot-accept/{i}");
let q = candidate_class(&s, s.as_bytes(), 0, c);
if check(&q).is_err() {
rejected += 1;
}
}
assert!(rejected < 15, "{name}: {rejected} of 60 rejected");
}
let p = generate_class("igneum-genesis", LoadClass::hot(96, 4));
let words = p.hot_words();
let loads = p.loads_per_hash();
let mut acc = Acc { sources: None, indices: None, sat_source: vec![0; 64], and_acc: [u32::MAX; 8], or_acc: [0; 8], saturated: 0, bit_ones: [0; 64], distinct_sum: 0 };
let mut la = vec![0u32; LANES * loads];
let mut buckets = [0u64; 16];
let mut hot_count = 0u64;
for (u, &b) in accept_base_nonces(&p.seed).iter().enumerate() {
run_unit(&p, u, b, &mut acc, &mut la).unwrap();
for &a in &la {
if a & 0xC000_0000 == 0x4000_0000 {
let idx = a & 0x3FFF_FFFF;
assert!(idx < words);
buckets[(idx as u64 * 16 / words as u64) as usize] += 1;
hot_count += 1;
}
}
}
assert_eq!(hot_count, 64 * 32 * 32, "32 hot loads per hash over 2,048 hashes");
let mean = hot_count as f64 / 16.0;
for (i, &b) in buckets.iter().enumerate() {
assert!((b as f64 - mean).abs() < 0.15 * mean, "bucket {i}: {b} against a mean of {mean}");
}
// the dataset distinct count still holds for the dataset loads alone
let r = check(&p).unwrap();
assert!(r.distinct_mean() > 120.0);
}
#[test]
fn base_nonces_are_aligned_and_seed_dependent() {
let a = accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 8]);
let b = accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 9]);
assert!(a.iter().all(|x| x & 31 == 0));
assert_ne!(a, b);
assert_eq!(a, accept_base_nonces(&[1, 2, 3, 4, 5, 6, 7, 8]));
}
#[test]
fn stale_load_detection_is_cyclic() {
let mut p = candidate("igneum-genesis", b"igneum-genesis", 0);
assert!(check_stale_loads(&p.instrs).is_ok(), "an accepted candidate has no stale load");
// Make the last instruction a load from r3 and the first a load from r3 with no write between (wrap).
let (first, last) = (0usize, INSTR_COUNT - 1);
p.instrs[last].op = Op::Load;
p.instrs[last].src = 3;
p.instrs[last].dst = 4;
p.instrs[first].op = Op::Load;
p.instrs[first].src = 3;
p.instrs[first].dst = 5;
assert_eq!(check_stale_loads(&p.instrs), Err(Reject::StaleLoadSource { instr: 0, reg: 3 }));
}
#[test]
fn injecting_write_detection() {
let mut p = candidate("igneum-genesis", b"igneum-genesis", 0);
for ins in p.instrs.iter_mut() {
if ins.dst == 6 && ins.op.injects() {
ins.op = Op::Rotl;
}
}
assert_eq!(check_injecting_writes(&p.instrs), Err(Reject::NoInjectingWrite { reg: 6 }));
}
/// The census's measured rates: about 5 percent of candidates rejected, 128 distinct loads for the rest.
#[test]
fn rejection_rate_and_distinct_loads_on_a_sample() {
let mut rejected = 0;
let mut dsum = 0.0;
let mut accepted = 0;
for i in 0..400u32 {
let s = format!("igneum-census-2026-10-03/{i}");
match check(&candidate(&s, s.as_bytes(), 0)) {
Ok(r) => {
accepted += 1;
dsum += r.distinct_mean();
assert!(r.distinct_mean() > 120.0);
}
Err(_) => rejected += 1,
}
}
assert!(rejected < 50, "{rejected} of 400 rejected");
assert!(dsum / accepted as f64 > 127.0, "mean distinct {}", dsum / accepted as f64);
}
/// The retired generator fails the rule on nearly every program (the census: 95 percent).
#[test]
fn v1_programs_are_mostly_rejected() {
let mut rejected = 0;
for i in 0..100u32 {
let s = format!("igneum-census-2026-10-03/{i}");
if check(&generate_v1(&s, &GeneratorConfig::default())).is_err() {
rejected += 1;
}
}
assert!(rejected > 80, "{rejected} of 100 rejected");
}
#[test]
fn generated_programs_pass() {
for s in ["igneum-genesis", "igneum-hourly", "igneum-second-seed"] {
assert!(check(&generate(s)).is_ok());
}
}
}

View file

@ -0,0 +1,234 @@
//! Header binding: the init words of the lottery hash commit to the block being mined.
//!
//! `docs/spec/01-lottery-hash.md` section 1.6 (open item O-1.9) proposes the rule implemented here:
//!
//! * The header nonce is 64 bits. Its low 32 bits are the lane nonce `n` (the per-thread nonce of the kernel).
//! * Its high 32 bits and the 256-bit pre-PoW header hash `H` form the init words:
//! `I = seed_words_from_bytes("igneum-block/" || H || nonce_hi_le32)`.
//! * `I` is a kernel argument, not a compile-time constant. The program (from the epoch seed) is compiled once per
//! epoch; `I` changes per block template.
//!
//! Choices the spec leaves open, fixed here (3 October 2026):
//!
//! | Choice | Rule | Why |
//! |---|---|---|
//! | `H` | the header hash with the nonce field set to zero and every other field as mined (rusty-kaspa `hash_override_nonce_time(header, 0, header.timestamp)`) | Kaspa's pre-PoW hash also zeroes the timestamp and absorbs it later in cSHAKE. The lane hash has no later step, so the timestamp must be inside `H` or a miner could reuse one nonce for many timestamps |
//! | 256-bit pow value | lane hash in the top 64 bits (little-endian bytes 24..32), low 192 bits zero | `pow <= target256` is then exactly `lane <= target256 >> 192` (section 1.10 candidate), so a GPU worker and the node compare the same 64-bit numbers; the block level is `leading_zeros(lane)` shifted |
//! | Day seed bytes (O-1.10 interim) | `"igneum-day/" || day_le64` with `day = header.timestamp_ms / 86,400,000` | The spec's proposal ties the day key to the first epoch seed of the day, which needs the VDF schedule. The interim rule keeps one 256 MiB cache per calendar day and needs no chain walk |
//! | Epoch seed bytes | the 32 bytes of the epoch block hash (devnet v0: the last selected-chain block below the epoch's start DAA score, genesis for epoch 0) | Section 1.12: `S_e = seed_words_from_bytes(program_seed_e)`; the VDF output replaces the block hash later without touching this crate |
//!
//! The packs' vectors (init words equal to the program seed) stay the conformance vectors for the generator,
//! interpreter and dataset. The bound vectors are in `README.md` and in the tests below (re-cut for generator
//! version 2 on 4 October 2026).
use crate::generator::LANES;
use crate::seed::seed_words_from_bytes;
use crate::verify::{interpret_warp_init, Epoch, WarpResult};
/// Domain tag of the init words.
pub const BLOCK_TAG: &[u8] = b"igneum-block/";
/// Domain tag of the interim day seed.
pub const DAY_TAG: &[u8] = b"igneum-day/";
/// Milliseconds per day, the clock of the interim day seed.
pub const DAY_MS: u64 = 86_400_000;
/// The lane nonce: low 32 bits of the header nonce.
#[inline]
pub fn lane_nonce(nonce: u64) -> u32 {
nonce as u32
}
/// The high 32 bits of the header nonce (the extra nonce that enters the init words).
#[inline]
pub fn nonce_hi(nonce: u64) -> u32 {
(nonce >> 32) as u32
}
/// `"igneum-block/" || H || nonce_hi_le32`, the bytes the init words are derived from.
pub fn block_init_bytes(header_prehash: &[u8; 32], nonce: u64) -> [u8; 49] {
let mut b = [0u8; 49];
b[..13].copy_from_slice(BLOCK_TAG);
b[13..45].copy_from_slice(header_prehash);
b[45..49].copy_from_slice(&nonce_hi(nonce).to_le_bytes());
b
}
/// The init words `I` for a header and a 64-bit nonce (only the high 32 bits of the nonce matter).
pub fn block_init_words(header_prehash: &[u8; 32], nonce: u64) -> [u32; 8] {
seed_words_from_bytes(&block_init_bytes(header_prehash, nonce))
}
/// Interim day seed bytes: `"igneum-day/" || day_le64`.
pub fn day_bytes(day_index: u64) -> [u8; 19] {
let mut b = [0u8; 19];
b[..11].copy_from_slice(DAY_TAG);
b[11..19].copy_from_slice(&day_index.to_le_bytes());
b
}
/// Day index of a header timestamp in milliseconds.
#[inline]
pub fn day_index(timestamp_ms: u64) -> u64 {
timestamp_ms / DAY_MS
}
/// The 256-bit pow value as little-endian bytes: the lane hash in bytes 24..32, zero elsewhere.
pub fn pow256_from_lane(lane: u64) -> [u8; 32] {
let mut b = [0u8; 32];
b[24..32].copy_from_slice(&lane.to_le_bytes());
b
}
/// The 64-bit target from a little-endian 256-bit target: its top 64 bits.
pub fn target64_from_le256(target: &[u8; 32]) -> u64 {
u64::from_le_bytes(target[24..32].try_into().unwrap())
}
/// Lower-case hex of bytes.
pub fn hex(bytes: &[u8]) -> String {
bytes.iter().map(|b| format!("{b:02x}")).collect()
}
/// Bytes from hex (either case). `None` on odd length or a bad digit.
pub fn unhex(s: &str) -> Option<Vec<u8>> {
if s.len() % 2 != 0 {
return None;
}
(0..s.len()).step_by(2).map(|i| u8::from_str_radix(&s[i..i + 2], 16).ok()).collect()
}
impl Epoch {
/// The 32 bound hashes of the aligned warp that contains `nonce`: lane `l` is the hash of
/// `(nonce_hi << 32) | ((lane_nonce & !31) + l)`.
pub fn hash_warp_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> [u64; LANES] {
self.interpret_warp_bound(header_prehash, nonce).hashes
}
pub fn interpret_warp_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> WarpResult {
let init = block_init_words(header_prehash, nonce);
interpret_warp_init(&self.program, &init, lane_nonce(nonce) & !31, &self.dataset)
}
/// The 32 bound hashes for already-derived init words (what a GPU worker computes per dispatch).
pub fn hash_warp_init(&self, init: &[u32; 8], base_lane_nonce: u32) -> [u64; LANES] {
interpret_warp_init(&self.program, init, base_lane_nonce, &self.dataset).hashes
}
/// The bound 64-bit lane hash of one header nonce.
pub fn hash_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> u64 {
self.hash_warp_bound(header_prehash, nonce)[(lane_nonce(nonce) & 31) as usize]
}
/// The bound 256-bit pow value (little-endian): the lane hash in the top 64 bits.
pub fn pow_bound(&self, header_prehash: &[u8; 32], nonce: u64) -> [u8; 32] {
pow256_from_lane(self.hash_bound(header_prehash, nonce))
}
/// `hash_bound(H, nonce) <= target64`.
pub fn verify_block_bound(&self, header_prehash: &[u8; 32], nonce: u64, target64: u64) -> bool {
self.hash_bound(header_prehash, nonce) <= target64
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::verify::DatasetMode;
use std::sync::OnceLock;
fn epoch() -> &'static Epoch {
static E: OnceLock<Epoch> = OnceLock::new();
E.get_or_init(|| Epoch::memory_hard("igneum-genesis", "2026-10-03"))
}
fn prehash_a() -> [u8; 32] {
[0u8; 32]
}
fn prehash_b() -> [u8; 32] {
let mut h = [0u8; 32];
for (i, b) in h.iter_mut().enumerate() {
*b = i as u8;
}
h
}
#[test]
fn init_bytes_layout() {
let b = block_init_bytes(&prehash_b(), 0x0000_0102_0000_0007);
assert_eq!(&b[..13], b"igneum-block/");
assert_eq!(&b[13..45], &prehash_b());
assert_eq!(&b[45..49], &[0x02, 0x01, 0x00, 0x00]);
assert_eq!(
block_init_words(&prehash_b(), 0x0000_0102_0000_0007),
block_init_words(&prehash_b(), 0x0000_0102_ffff_ffff)
);
assert_ne!(block_init_words(&prehash_b(), 0), block_init_words(&prehash_b(), 1 << 32));
}
#[test]
fn day_bytes_layout() {
let b = day_bytes(20_729);
assert_eq!(&b[..11], b"igneum-day/");
assert_eq!(&b[11..], &20_729u64.to_le_bytes());
assert_eq!(day_index(0x1a0ff0f7c00), 20_729);
}
#[test]
fn pow256_and_target64() {
let p = pow256_from_lane(0x0123_4567_89ab_cdef);
assert_eq!(&p[..24], &[0u8; 24]);
assert_eq!(target64_from_le256(&p), 0x0123_4567_89ab_cdef);
assert_eq!(hex(&p[24..]), "efcdab8967452301");
assert_eq!(unhex("efcdab8967452301").unwrap(), p[24..].to_vec());
assert!(unhex("abc").is_none());
}
/// The bound form is a different hash from the unbound one, depends on H and on the high nonce bits, and the
/// single-nonce form agrees with the warp form.
#[test]
fn bound_hash_properties() {
let e = epoch();
let a = prehash_a();
let b = prehash_b();
assert_ne!(e.hash_bound(&a, 0), e.hash(0), "bound differs from the pack vector");
assert_ne!(e.hash_bound(&a, 0), e.hash_bound(&b, 0), "H enters the hash");
assert_ne!(e.hash_bound(&a, 0), e.hash_bound(&a, 1 << 32), "nonce_hi enters the hash");
let w = e.hash_warp_bound(&a, (1 << 32) | 37);
assert_eq!(w[5], e.hash_bound(&a, (1 << 32) | 37));
assert_eq!(w[0], e.hash_bound(&a, 1 << 32 | 32));
let init = block_init_words(&a, 1 << 32);
assert_eq!(e.hash_warp_init(&init, 32), w);
let t = e.hash_bound(&a, 7);
assert!(e.verify_block_bound(&a, 7, t));
assert!(!e.verify_block_bound(&a, 7, t - 1));
assert_eq!(e.pow_bound(&a, 7), pow256_from_lane(t));
}
/// The 8 bound vectors printed in README.md (seed igneum-genesis, day 2026-10-03, memory-hard, 2^28 words,
/// generator v2 since 4 October 2026).
#[test]
fn bound_vectors() {
let e = epoch();
let a = prehash_a();
let b = prehash_b();
let cases: [(&[u8; 32], u64, u64); 8] = [
(&a, 0, 0x746c567b090acf6a),
(&a, 1, 0x45a619f860880c73),
(&a, 31, 0x9aa495e43dedbfe6),
(&a, 4096, 0x2e6ffd7624d3cba2),
(&a, 1 << 32, 0x38a5cea1fb01431a),
(&b, 0, 0x2a79c5e4797bf6aa),
(&b, (1 << 32) | 5, 0xa243e0c61aa1b82e),
(&b, u64::MAX, 0x9c7bbfbd064fe1a4),
];
for (h, nonce, want) in cases {
assert_eq!(e.hash_bound(h, nonce), want, "H {} nonce {nonce}", hex(h));
}
}
#[test]
fn closed_form_bound_also_works() {
let e = Epoch::new("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 28);
assert_ne!(e.hash_bound(&prehash_a(), 0), e.hash(0));
}
}

View file

@ -0,0 +1,152 @@
//! BLAKE2b (RFC 7693), the chain's own hash family (spec 01 section 0.6), written out here so the crate keeps its
//! rule of no dependency outside the standard library. Used by class v5's state leaves (`crate::state`):
//! `blake2b_512` for a leaf digest, `blake2b_256` for the sample order. Unkeyed, no salt, no personalisation.
//! Checked against the RFC's "abc" vector and the empty-input vector in the tests.
const IV: [u64; 8] = [
0x6a09e667f3bcc908,
0xbb67ae8584caa73b,
0x3c6ef372fe94f82b,
0xa54ff53a5f1d36f1,
0x510e527fade682d1,
0x9b05688c2b3e6c1f,
0x1f83d9abfb41bd6b,
0x5be0cd19137e2179,
];
const SIGMA: [[usize; 16]; 12] = [
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15],
[14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3],
[11, 8, 12, 0, 5, 2, 15, 13, 10, 14, 3, 6, 7, 1, 9, 4],
[7, 9, 3, 1, 13, 12, 11, 14, 2, 6, 5, 10, 4, 0, 15, 8],
[9, 0, 5, 7, 2, 4, 10, 15, 14, 1, 11, 12, 6, 8, 3, 13],
[2, 12, 6, 10, 0, 11, 8, 3, 4, 13, 7, 5, 15, 14, 1, 9],
[12, 5, 1, 15, 14, 13, 4, 10, 0, 7, 6, 3, 9, 2, 8, 11],
[13, 11, 7, 14, 12, 1, 3, 9, 5, 0, 15, 4, 8, 6, 2, 10],
[6, 15, 14, 9, 11, 3, 0, 8, 12, 2, 13, 7, 1, 4, 10, 5],
[10, 2, 8, 4, 7, 6, 1, 5, 15, 11, 9, 14, 3, 12, 13, 0],
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15],
[14, 10, 4, 8, 9, 15, 13, 6, 1, 12, 0, 2, 11, 7, 5, 3],
];
#[inline(always)]
fn g(v: &mut [u64; 16], a: usize, b: usize, c: usize, d: usize, x: u64, y: u64) {
v[a] = v[a].wrapping_add(v[b]).wrapping_add(x);
v[d] = (v[d] ^ v[a]).rotate_right(32);
v[c] = v[c].wrapping_add(v[d]);
v[b] = (v[b] ^ v[c]).rotate_right(24);
v[a] = v[a].wrapping_add(v[b]).wrapping_add(y);
v[d] = (v[d] ^ v[a]).rotate_right(16);
v[c] = v[c].wrapping_add(v[d]);
v[b] = (v[b] ^ v[c]).rotate_right(63);
}
fn compress(h: &mut [u64; 8], block: &[u8; 128], t: u128, last: bool) {
let mut m = [0u64; 16];
for (i, w) in m.iter_mut().enumerate() {
*w = u64::from_le_bytes(block[i * 8..i * 8 + 8].try_into().unwrap());
}
let mut v = [0u64; 16];
v[..8].copy_from_slice(h);
v[8..].copy_from_slice(&IV);
v[12] ^= t as u64;
v[13] ^= (t >> 64) as u64;
if last {
v[14] = !v[14];
}
for s in SIGMA.iter() {
g(&mut v, 0, 4, 8, 12, m[s[0]], m[s[1]]);
g(&mut v, 1, 5, 9, 13, m[s[2]], m[s[3]]);
g(&mut v, 2, 6, 10, 14, m[s[4]], m[s[5]]);
g(&mut v, 3, 7, 11, 15, m[s[6]], m[s[7]]);
g(&mut v, 0, 5, 10, 15, m[s[8]], m[s[9]]);
g(&mut v, 1, 6, 11, 12, m[s[10]], m[s[11]]);
g(&mut v, 2, 7, 8, 13, m[s[12]], m[s[13]]);
g(&mut v, 3, 4, 9, 14, m[s[14]], m[s[15]]);
}
for i in 0..8 {
h[i] ^= v[i] ^ v[i + 8];
}
}
/// Unkeyed BLAKE2b of `data` with an output of `out_len` bytes (1..=64), written into `out[..out_len]`.
pub fn blake2b(out: &mut [u8], out_len: usize, data: &[u8]) {
assert!((1..=64).contains(&out_len) && out.len() >= out_len);
let mut h = IV;
h[0] ^= 0x0101_0000 ^ out_len as u64;
let mut t: u128 = 0;
let n = data.len();
// every full block but the last; the last block (possibly empty) is compressed with the final flag
let full = if n == 0 { 0 } else { (n - 1) / 128 };
for i in 0..full {
let block: &[u8; 128] = data[i * 128..i * 128 + 128].try_into().unwrap();
t += 128;
compress(&mut h, block, t, false);
}
let mut last = [0u8; 128];
let rest = &data[full * 128..];
last[..rest.len()].copy_from_slice(rest);
t += rest.len() as u128;
compress(&mut h, &last, t, true);
let mut bytes = [0u8; 64];
for (i, w) in h.iter().enumerate() {
bytes[i * 8..i * 8 + 8].copy_from_slice(&w.to_le_bytes());
}
out[..out_len].copy_from_slice(&bytes[..out_len]);
}
/// BLAKE2b-512 of the concatenation of `parts`.
pub fn blake2b_512(parts: &[&[u8]]) -> [u8; 64] {
let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum());
for p in parts {
data.extend_from_slice(p);
}
let mut out = [0u8; 64];
blake2b(&mut out, 64, &data);
out
}
/// BLAKE2b-256 of the concatenation of `parts`.
pub fn blake2b_256(parts: &[&[u8]]) -> [u8; 32] {
let mut data = Vec::with_capacity(parts.iter().map(|p| p.len()).sum());
for p in parts {
data.extend_from_slice(p);
}
let mut out = [0u8; 32];
blake2b(&mut out, 32, &data);
out
}
#[cfg(test)]
mod tests {
use super::*;
fn hex(b: &[u8]) -> String {
b.iter().map(|x| format!("{x:02x}")).collect()
}
/// RFC 7693 appendix A ("abc"), the empty input, and a two-block input against the reference implementation's
/// known values (the three-block "The quick brown fox" vector of the BLAKE2 test suite).
#[test]
fn rfc_7693_vectors() {
assert_eq!(
hex(&blake2b_512(&[b"abc"])),
"ba80a53f981c4d0d6a2797b69f12f6e94c212f14685ac4b74b12bb6fdbffa2d17d87c5392aab792dc252d5de4533cc9518d38aa8dbf1925ab92386edd4009923"
);
assert_eq!(
hex(&blake2b_512(&[b""])),
"786a02f742015903c6c6fd852552d272912f4740e15847618a86e217f71f5419d25e1031afee585313896444934eb04b903a685b1448b755d56f701afe9be2ce"
);
assert_eq!(hex(&blake2b_256(&[b"abc"])), "bddd813c634239723171ef3fee98579b94964e3bb1cb3e427262c8c068d52319");
assert_eq!(hex(&blake2b_256(&[b""])), "0e5751c026e543b2e8ab2eb06099daa1d1e5df47778f7787faab45cdf12fe3a8");
// a 128-byte input is exactly one full block compressed as the last; 129 bytes takes two
let one = [0x61u8; 128];
let two = [0x61u8; 129];
assert_ne!(blake2b_512(&[&one]), blake2b_512(&[&two]));
assert_eq!(blake2b_512(&[&one[..64], &one[64..]]), blake2b_512(&[&one]), "parts concatenate");
assert_eq!(
hex(&blake2b_512(&[b"The quick brown fox jumps over the lazy dog"])),
"a8add4bdddfd93e4877d2746e62817b116364a1fa7bc148d95090bc7333b3673f82401cf7aa2e4cb1ecd90296e3f14cb5413f8ed77be73045b13914cdcd6a918"
);
}
}

View file

@ -0,0 +1,978 @@
//! The per-day item-derivation program (Counter ASIC 3.0 item 2, `docs/plans/counter-asic-3-derivation.md`):
//! RandomX's SuperscalarHash idea (`vendor/RandomX/src/superscalar.cpp`, read 6 October 2026 at commit 7607fb2)
//! rebuilt for a 16-word item on a GPU. In place of the fixed-shape mixer `M_r` of spec 01 section 1.8.4, each of
//! the nine mixer slots of an item (one before each of the 8 dependent cache reads, one after the last) runs a
//! straight-line program of [`DERIVE_LEN`] instructions drawn once a day from the day key stream, from a fixed
//! set of twelve two-register forms. The 8 dependent cache reads per item are untouched.
//!
//! Rules of the draw (the dependency chain of SuperscalarHash, made strict):
//! * every instruction reads the chain register `c`, the register the previous instruction wrote (`s[0]`, the
//! address word, at the start of each round program), and writes a register `d != c`, which becomes the chain;
//! so no two instructions of a program can run in parallel, and no two consecutive instructions write one
//! register (the "ror r,C1; ror r,C2" and "xor r,r2; xor r,r2" merges of SuperscalarHash's `selectDestination`
//! cannot arise);
//! * every form is a bijection on the 16-word state (the old `d` enters through `+=`, `-=`, `^=`, an odd multiply,
//! or a rotation of itself), so a program loses no entropy, the property `M_r` has;
//! * the forms are integer only, modulo 2^32, with rotations by 1..31: bit-exact on Metal, CUDA and OpenCL by the
//! same argument as the lottery hash's families (spec 01 section 1.14); no division, no float, no branch;
//! * four draws per instruction in a fixed order, so the stream position of every draw is fixed by the index.
//!
//! The acceptance test ([`DeriveProgram::check`]) rejects a degenerate draw and the next attempt is drawn from the
//! continuation of the stream, the rule the program generator uses (spec 01 section 1.4.6).
use crate::memhard::ITEM_ROUNDS;
use crate::seed::SplitMix64;
/// Registers of the item state (the item is 16 words).
pub const DERIVE_REGS: usize = 16;
/// Round programs per item: one before each cache read and one after the last (`ITEM_ROUNDS + 1`).
pub const DERIVE_PROGRAMS: usize = ITEM_ROUNDS + 1;
/// Instructions per round program for the x8-equivalent operation count (the candidate, class "dr736"): 9 x 736
/// = 6,624 instructions per item at a mean of 1.62 GPU operations (1.52 chip operations) each, about 10,730 GPU
/// operations, 10,070 chip operations and 1,460 multiplies per item. The x8 mixer, counted from the code
/// (`memhard::mixer`, 16 x (xor, add, mul) + 8 quarter rounds x 12 = 144 operations as written, 128 with the
/// `RC + rk` adds hoisted as constants, 16 multiplies; `chip-model-v3.md` section 1 prices 130 from the spec text):
/// 72 applications = 10,368 as written, 9,216 hoisted, 1,152 multiplies. The floors below are those three.
pub const DERIVE_LEN_X8: u32 = 736;
/// Draws per instruction: the op roll, the destination roll, the second-source roll and the immediate.
pub const DRAWS_PER_INSTR: u64 = 4;
/// The floor of chip operations per item (the x8 mixer with its constants hoisted: 72 x 128).
pub const OPS_FLOOR_X8: u64 = 9_216;
/// The floor of GPU operations per item (the x8 mixer as written: 72 x 144).
pub const GPU_OPS_FLOOR_X8: u64 = 10_368;
/// The floor of multiplies per item (the x8 mixer's 72 x 16).
pub const MULS_FLOOR_X8: u64 = 1_152;
/// Distinct rotation amounts an item's programs must use, at least.
pub const DISTINCT_ROTS_FLOOR: usize = 8;
/// Attempts before the generator gives up (never reached: see [`DeriveProgram::draw`]).
pub const MAX_ATTEMPTS: u32 = 64;
/// The twelve forms. `c` is the chain register (the previous destination), `d` the destination (`d != c`), `b` a
/// third register (`b != d`, `b != c`), `k` a rotation in 1..31, `i` a 32-bit constant (odd for `MulC`).
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
#[repr(u8)]
pub enum DOp {
/// `d += c`
Add = 0,
/// `d -= c`
Sub = 1,
/// `d ^= c`
Xor = 2,
/// `d *= (c OR 1)`: the multiply-lo form, odd so it is a bijection on `d`
Mul = 3,
/// `d = rotl(d, k) + c`
Rot = 4,
/// `d = rotl(d ^ c, k)`
XRot = 5,
/// `d += c + i`
AddC = 6,
/// `d ^= c ^ i`
XorC = 7,
/// `d = (d ^ c) * i`, `i` odd: the per-word form of `M_r` with the chain in place of the round constant
MulC = 8,
/// `d = d * i + c`, `i` odd
MulC2 = 9,
/// `d ^= (c AND b)`
AndX = 10,
/// `d += (c OR b)`
OrX = 11,
}
/// The op weights in percent, in draw order (sum 100). Fixed at genesis; only the order, the registers and the
/// constants are drawn.
pub const DOP_WEIGHTS: [(DOp, u64); 12] = [
(DOp::Add, 14),
(DOp::Sub, 10),
(DOp::Xor, 14),
(DOp::Mul, 10),
(DOp::Rot, 10),
(DOp::XRot, 10),
(DOp::AddC, 6),
(DOp::XorC, 6),
(DOp::MulC, 8),
(DOp::MulC2, 4),
(DOp::AndX, 4),
(DOp::OrX, 4),
];
impl DOp {
pub fn from_u8(v: u8) -> Option<DOp> {
DOP_WEIGHTS.iter().map(|(o, _)| *o).find(|o| *o as u8 == v)
}
pub fn name(self) -> &'static str {
match self {
DOp::Add => "add",
DOp::Sub => "sub",
DOp::Xor => "xor",
DOp::Mul => "mul",
DOp::Rot => "rot",
DOp::XRot => "xrot",
DOp::AddC => "addc",
DOp::XorC => "xorc",
DOp::MulC => "mulc",
DOp::MulC2 => "mulc2",
DOp::AndX => "andx",
DOp::OrX => "orx",
}
}
/// Integer operations as a GPU executes the form (every `|`, `&`, `+`, `^`, `*`, rotate counts one).
pub fn gpu_ops(self) -> u64 {
match self {
DOp::Add | DOp::Sub | DOp::Xor => 1,
_ => 2,
}
}
/// Integer operations as the chip model counts them (`c OR 1` is a wire on a chip, so `Mul` is one multiply;
/// a constant folded into a chain value is still an add or an xor, so every other two-op form stays two).
pub fn chip_ops(self) -> u64 {
match self {
DOp::Add | DOp::Sub | DOp::Xor | DOp::Mul => 1,
_ => 2,
}
}
pub fn is_mul(self) -> bool {
matches!(self, DOp::Mul | DOp::MulC | DOp::MulC2)
}
pub fn has_rot(self) -> bool {
matches!(self, DOp::Rot | DOp::XRot)
}
pub fn has_third(self) -> bool {
matches!(self, DOp::AndX | DOp::OrX)
}
pub fn has_imm(self) -> bool {
matches!(self, DOp::AddC | DOp::XorC | DOp::MulC | DOp::MulC2)
}
/// The op of a roll in 0..99.
pub fn for_roll(roll: u64) -> DOp {
let mut acc = 0u64;
for (op, w) in DOP_WEIGHTS {
acc += w;
if roll < acc {
return op;
}
}
DOp::OrX
}
}
/// One instruction. `src` is the chain register (carried so the interpreter and the emitter need no state).
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
pub struct DInstr {
pub op: DOp,
pub dst: u8,
pub src: u8,
/// The third register of `AndX` and `OrX`; 0 on every other form (drawn and unused).
pub src2: u8,
/// The rotation 1..31 of `Rot` and `XRot`, else 0.
pub rot: u8,
/// The constant of `AddC`, `XorC` (any), `MulC` and `MulC2` (odd); 0 on every other form.
pub imm: u32,
}
/// The nine round programs of an item for one day, with the attempt that passed the acceptance test.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct DeriveProgram {
pub len: u32,
pub attempt: u32,
pub rounds: Vec<Vec<DInstr>>,
}
/// Why a candidate was rejected.
#[derive(Clone, Debug, PartialEq, Eq)]
pub enum DeriveReject {
/// A register no instruction of round program `round` writes.
RegisterNeverWritten { round: usize, reg: u8 },
/// Fewer than [`DISTINCT_ROTS_FLOOR`] distinct rotation amounts over the item's programs.
RotationsDegenerate { distinct: usize },
/// Chip operations per item under [`OPS_FLOOR_X8`] scaled to the length.
OpsUnderFloor { ops: u64, floor: u64 },
/// GPU operations per item under [`GPU_OPS_FLOOR_X8`] scaled to the length.
GpuOpsUnderFloor { ops: u64, floor: u64 },
/// Multiplies per item under [`MULS_FLOOR_X8`] scaled to the length.
MulsUnderFloor { muls: u64, floor: u64 },
}
impl std::fmt::Display for DeriveReject {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
DeriveReject::RegisterNeverWritten { round, reg } => write!(f, "register {reg} never written in round program {round}"),
DeriveReject::RotationsDegenerate { distinct } => write!(f, "only {distinct} distinct rotation amounts"),
DeriveReject::OpsUnderFloor { ops, floor } => write!(f, "{ops} chip operations per item, floor {floor}"),
DeriveReject::GpuOpsUnderFloor { ops, floor } => write!(f, "{ops} GPU operations per item, floor {floor}"),
DeriveReject::MulsUnderFloor { muls, floor } => write!(f, "{muls} multiplies per item, floor {floor}"),
}
}
}
impl DeriveProgram {
/// Draw one candidate of `len` instructions per round program from `rng` (four draws per instruction).
pub fn draw_candidate(rng: &mut SplitMix64, len: u32, attempt: u32) -> DeriveProgram {
let mut rounds = Vec::with_capacity(DERIVE_PROGRAMS);
for _ in 0..DERIVE_PROGRAMS {
let mut prog = Vec::with_capacity(len as usize);
let mut chain = 0u8;
for _ in 0..len {
let op = DOp::for_roll(rng.below(100));
// the destination: the 15 registers other than the chain, in ascending order
let d_roll = rng.below((DERIVE_REGS - 1) as u64) as u8;
let dst = if d_roll >= chain { d_roll + 1 } else { d_roll };
// the third register: the 14 registers other than dst and the chain, in ascending order
let b_roll = rng.below((DERIVE_REGS - 2) as u64) as u8;
let (lo, hi) = if dst < chain { (dst, chain) } else { (chain, dst) };
let mut b = b_roll;
if b >= lo {
b += 1;
}
if b >= hi {
b += 1;
}
let x = rng.next() as u32;
let mut ins = DInstr { op, dst, src: chain, src2: 0, rot: 0, imm: 0 };
if op.has_third() {
ins.src2 = b;
}
if op.has_rot() {
ins.rot = 1 + (x % 31) as u8;
}
if op.has_imm() {
ins.imm = if matches!(op, DOp::MulC | DOp::MulC2) { x | 1 } else { x };
}
prog.push(ins);
chain = dst;
}
rounds.push(prog);
}
DeriveProgram { len, attempt, rounds }
}
/// Draw the program of a day: candidates from `rng` in turn until one passes [`DeriveProgram::check`].
/// Panics after [`MAX_ATTEMPTS`] (the floors sit more than 7 standard deviations under the expected counts, so
/// a rejection is a rare event and 64 in a row is not one that happens).
pub fn draw(rng: &mut SplitMix64, len: u32) -> DeriveProgram {
for attempt in 0..MAX_ATTEMPTS {
let p = Self::draw_candidate(rng, len, attempt);
if p.check().is_ok() {
return p;
}
}
panic!("derivation program: {MAX_ATTEMPTS} candidates rejected in a row");
}
/// The floors for this length (chip operations, GPU operations, multiplies): the x8 floors scaled by
/// `len / DERIVE_LEN_X8`, so a shorter class, measured as a fallback, has its own proportional floors.
pub fn floors(len: u32) -> (u64, u64, u64) {
let scale = |f: u64| f * len as u64 / DERIVE_LEN_X8 as u64;
(scale(OPS_FLOOR_X8), scale(GPU_OPS_FLOOR_X8), scale(MULS_FLOOR_X8))
}
/// The acceptance test: every register written in every round program; at least [`DISTINCT_ROTS_FLOOR`]
/// distinct rotation amounts; chip operations, GPU operations and multiplies per item at or above the floors
/// (the x8 mixer's counts from the code). The structural
/// rules (`dst != src`, the third register distinct, rotations in 1..31, odd multiplier constants) hold by
/// construction and are asserted.
pub fn check(&self) -> Result<(), DeriveReject> {
let mut rots = [false; 32];
for (r, prog) in self.rounds.iter().enumerate() {
let mut written = [false; DERIVE_REGS];
let mut chain = 0u8;
for ins in prog {
assert!(ins.src == chain && ins.dst != ins.src && (ins.dst as usize) < DERIVE_REGS, "chain rule");
if ins.op.has_third() {
assert!(ins.src2 != ins.dst && ins.src2 != ins.src && (ins.src2 as usize) < DERIVE_REGS, "third register");
}
if ins.op.has_rot() {
assert!((1..=31).contains(&ins.rot), "rotation");
rots[ins.rot as usize] = true;
}
if matches!(ins.op, DOp::MulC | DOp::MulC2) {
assert!(ins.imm & 1 == 1, "odd multiplier");
}
written[ins.dst as usize] = true;
chain = ins.dst;
}
if let Some(reg) = written.iter().position(|w| !w) {
return Err(DeriveReject::RegisterNeverWritten { round: r, reg: reg as u8 });
}
}
let distinct = rots.iter().filter(|r| **r).count();
if distinct < DISTINCT_ROTS_FLOOR {
return Err(DeriveReject::RotationsDegenerate { distinct });
}
let (ops_floor, gpu_floor, muls_floor) = Self::floors(self.len);
let ops = self.chip_ops();
if ops < ops_floor {
return Err(DeriveReject::OpsUnderFloor { ops, floor: ops_floor });
}
let gpu = self.gpu_ops();
if gpu < gpu_floor {
return Err(DeriveReject::GpuOpsUnderFloor { ops: gpu, floor: gpu_floor });
}
let muls = self.muls();
if muls < muls_floor {
return Err(DeriveReject::MulsUnderFloor { muls, floor: muls_floor });
}
Ok(())
}
pub fn instr_count(&self) -> u64 {
self.rounds.iter().map(|p| p.len() as u64).sum()
}
pub fn gpu_ops(&self) -> u64 {
self.rounds.iter().flatten().map(|i| i.op.gpu_ops()).sum()
}
pub fn chip_ops(&self) -> u64 {
self.rounds.iter().flatten().map(|i| i.op.chip_ops()).sum()
}
pub fn muls(&self) -> u64 {
self.rounds.iter().flatten().filter(|i| i.op.is_mul()).count() as u64
}
/// Count per op, in [`DOP_WEIGHTS`] order.
pub fn op_counts(&self) -> [u64; 12] {
let mut c = [0u64; 12];
for i in self.rounds.iter().flatten() {
c[i.op as usize] += 1;
}
c
}
/// "add=887 sub=..." in weight order.
pub fn op_mix(&self) -> String {
let c = self.op_counts();
DOP_WEIGHTS.iter().map(|(o, _)| format!("{}={}", o.name(), c[*o as usize])).collect::<Vec<_>>().join(" ")
}
/// FNV-1a 64 over the instruction stream (op, dst, src, src2, rot, imm as bytes): the program's fingerprint
/// for packs and logs.
pub fn fingerprint(&self) -> u64 {
let mut b = Vec::with_capacity(self.instr_count() as usize * 9);
for i in self.rounds.iter().flatten() {
b.push(i.op as u8);
b.push(i.dst);
b.push(i.src);
b.push(i.src2);
b.push(i.rot);
b.extend_from_slice(&i.imm.to_le_bytes());
}
crate::seed::fnv1a64(&b)
}
}
/// Lanes of the SoA interpreter: the verifier derives up to 32 distinct items per load (one per lane of the
/// unit), so each instruction runs across 32 item states at once and the dispatch is paid once per 32 items.
pub const SOA_LANES: usize = 32;
/// The item states of a batch, word-major: `st[reg][lane]`.
pub type SoaState = [[u32; SOA_LANES]; DERIVE_REGS];
#[inline(always)]
fn rotl(x: u32, n: u32) -> u32 {
x.rotate_left(n)
}
/// The twelve forms over a batch, one function each, every one a straight loop over the lanes the compiler
/// vectorises. The destination row and the source rows are distinct by the chain rule (`dst != src`, and the third
/// register distinct from both: asserted by [`DeriveProgram::check`] and checked here in debug builds), so the
/// rows are addressed through raw pointers rather than copied out of the state.
mod forms {
use super::{rotl, DInstr, SoaState, SOA_LANES};
#[inline(always)]
pub fn add(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src);
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
*d = d.wrapping_add(*c);
}
}
}
#[inline(always)]
pub fn sub(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src);
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
*d = d.wrapping_sub(*c);
}
}
}
#[inline(always)]
pub fn xor(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src);
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
*d ^= *c;
}
}
}
#[inline(always)]
pub fn mul(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src);
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
*d = d.wrapping_mul(*c | 1);
}
}
}
#[inline(always)]
pub fn rot(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src);
let r = ins.rot as u32;
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
*d = rotl(*d, r).wrapping_add(*c);
}
}
}
#[inline(always)]
pub fn xrot(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src);
let r = ins.rot as u32;
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
*d = rotl(*d ^ *c, r);
}
}
}
#[inline(always)]
pub fn addc(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src);
let i = ins.imm;
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
*d = d.wrapping_add(c.wrapping_add(i));
}
}
}
#[inline(always)]
pub fn xorc(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src);
let i = ins.imm;
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
*d ^= *c ^ i;
}
}
}
#[inline(always)]
pub fn mulc(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src);
let i = ins.imm;
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
*d = (*d ^ *c).wrapping_mul(i);
}
}
}
#[inline(always)]
pub fn mulc2(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src);
let i = ins.imm;
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
*d = d.wrapping_mul(i).wrapping_add(*c);
}
}
}
#[inline(always)]
pub fn andx(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src && ins.src2 != ins.dst && ins.src2 != ins.src);
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
let bp = st.as_ptr().add(ins.src2 as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
let b = &*bp.add(k);
*d ^= *c & *b;
}
}
}
#[inline(always)]
pub fn orx(ins: &DInstr, st: &mut SoaState) {
debug_assert!(ins.dst != ins.src && ins.src2 != ins.dst && ins.src2 != ins.src);
// SAFETY: dst, src (and src2) are distinct registers below DERIVE_REGS, so the rows do not alias and the
// pointers stay inside `st`.
unsafe {
let dp = st.as_mut_ptr().add(ins.dst as usize) as *mut u32;
let cp = st.as_ptr().add(ins.src as usize) as *const u32;
let bp = st.as_ptr().add(ins.src2 as usize) as *const u32;
for k in 0..SOA_LANES {
let d = &mut *dp.add(k);
let c = &*cp.add(k);
let b = &*bp.add(k);
*d = d.wrapping_add(*c | *b);
}
}
}
}
/// One instruction over the batch (the single dispatch; [`run_round`] dispatches on pairs).
#[inline(always)]
pub fn run_instr(ins: &DInstr, st: &mut SoaState) {
match ins.op {
DOp::Add => forms::add(ins, st),
DOp::Sub => forms::sub(ins, st),
DOp::Xor => forms::xor(ins, st),
DOp::Mul => forms::mul(ins, st),
DOp::Rot => forms::rot(ins, st),
DOp::XRot => forms::xrot(ins, st),
DOp::AddC => forms::addc(ins, st),
DOp::XorC => forms::xorc(ins, st),
DOp::MulC => forms::mulc(ins, st),
DOp::MulC2 => forms::mulc2(ins, st),
DOp::AndX => forms::andx(ins, st),
DOp::OrX => forms::orx(ins, st),
}
}
/// Run one round program over the batch. The dispatch is on PAIRS of instructions (144 arms, one indirect branch
/// per two instructions): the op sequence of a drawn program is random, so the branch predictor misses most
/// dispatches, and the miss (about 3.7 ns of the 7.2 ns an instruction cost per batch on one M5 Max core, measured
/// with `examples/derive_perf.rs` on 6 October 2026, a functional run) is paid once per pair instead of once per
/// instruction. The result is bit for bit that of [`run_instr`] in sequence.
#[inline(never)]
pub fn run_round(prog: &[DInstr], st: &mut SoaState) {
let mut it = prog.chunks_exact(2);
for pair in &mut it {
let (a, b) = (&pair[0], &pair[1]);
match (a.op as u8) * 12 + b.op as u8 {
0 => { forms::add(a, st); forms::add(b, st); }
1 => { forms::add(a, st); forms::sub(b, st); }
2 => { forms::add(a, st); forms::xor(b, st); }
3 => { forms::add(a, st); forms::mul(b, st); }
4 => { forms::add(a, st); forms::rot(b, st); }
5 => { forms::add(a, st); forms::xrot(b, st); }
6 => { forms::add(a, st); forms::addc(b, st); }
7 => { forms::add(a, st); forms::xorc(b, st); }
8 => { forms::add(a, st); forms::mulc(b, st); }
9 => { forms::add(a, st); forms::mulc2(b, st); }
10 => { forms::add(a, st); forms::andx(b, st); }
11 => { forms::add(a, st); forms::orx(b, st); }
12 => { forms::sub(a, st); forms::add(b, st); }
13 => { forms::sub(a, st); forms::sub(b, st); }
14 => { forms::sub(a, st); forms::xor(b, st); }
15 => { forms::sub(a, st); forms::mul(b, st); }
16 => { forms::sub(a, st); forms::rot(b, st); }
17 => { forms::sub(a, st); forms::xrot(b, st); }
18 => { forms::sub(a, st); forms::addc(b, st); }
19 => { forms::sub(a, st); forms::xorc(b, st); }
20 => { forms::sub(a, st); forms::mulc(b, st); }
21 => { forms::sub(a, st); forms::mulc2(b, st); }
22 => { forms::sub(a, st); forms::andx(b, st); }
23 => { forms::sub(a, st); forms::orx(b, st); }
24 => { forms::xor(a, st); forms::add(b, st); }
25 => { forms::xor(a, st); forms::sub(b, st); }
26 => { forms::xor(a, st); forms::xor(b, st); }
27 => { forms::xor(a, st); forms::mul(b, st); }
28 => { forms::xor(a, st); forms::rot(b, st); }
29 => { forms::xor(a, st); forms::xrot(b, st); }
30 => { forms::xor(a, st); forms::addc(b, st); }
31 => { forms::xor(a, st); forms::xorc(b, st); }
32 => { forms::xor(a, st); forms::mulc(b, st); }
33 => { forms::xor(a, st); forms::mulc2(b, st); }
34 => { forms::xor(a, st); forms::andx(b, st); }
35 => { forms::xor(a, st); forms::orx(b, st); }
36 => { forms::mul(a, st); forms::add(b, st); }
37 => { forms::mul(a, st); forms::sub(b, st); }
38 => { forms::mul(a, st); forms::xor(b, st); }
39 => { forms::mul(a, st); forms::mul(b, st); }
40 => { forms::mul(a, st); forms::rot(b, st); }
41 => { forms::mul(a, st); forms::xrot(b, st); }
42 => { forms::mul(a, st); forms::addc(b, st); }
43 => { forms::mul(a, st); forms::xorc(b, st); }
44 => { forms::mul(a, st); forms::mulc(b, st); }
45 => { forms::mul(a, st); forms::mulc2(b, st); }
46 => { forms::mul(a, st); forms::andx(b, st); }
47 => { forms::mul(a, st); forms::orx(b, st); }
48 => { forms::rot(a, st); forms::add(b, st); }
49 => { forms::rot(a, st); forms::sub(b, st); }
50 => { forms::rot(a, st); forms::xor(b, st); }
51 => { forms::rot(a, st); forms::mul(b, st); }
52 => { forms::rot(a, st); forms::rot(b, st); }
53 => { forms::rot(a, st); forms::xrot(b, st); }
54 => { forms::rot(a, st); forms::addc(b, st); }
55 => { forms::rot(a, st); forms::xorc(b, st); }
56 => { forms::rot(a, st); forms::mulc(b, st); }
57 => { forms::rot(a, st); forms::mulc2(b, st); }
58 => { forms::rot(a, st); forms::andx(b, st); }
59 => { forms::rot(a, st); forms::orx(b, st); }
60 => { forms::xrot(a, st); forms::add(b, st); }
61 => { forms::xrot(a, st); forms::sub(b, st); }
62 => { forms::xrot(a, st); forms::xor(b, st); }
63 => { forms::xrot(a, st); forms::mul(b, st); }
64 => { forms::xrot(a, st); forms::rot(b, st); }
65 => { forms::xrot(a, st); forms::xrot(b, st); }
66 => { forms::xrot(a, st); forms::addc(b, st); }
67 => { forms::xrot(a, st); forms::xorc(b, st); }
68 => { forms::xrot(a, st); forms::mulc(b, st); }
69 => { forms::xrot(a, st); forms::mulc2(b, st); }
70 => { forms::xrot(a, st); forms::andx(b, st); }
71 => { forms::xrot(a, st); forms::orx(b, st); }
72 => { forms::addc(a, st); forms::add(b, st); }
73 => { forms::addc(a, st); forms::sub(b, st); }
74 => { forms::addc(a, st); forms::xor(b, st); }
75 => { forms::addc(a, st); forms::mul(b, st); }
76 => { forms::addc(a, st); forms::rot(b, st); }
77 => { forms::addc(a, st); forms::xrot(b, st); }
78 => { forms::addc(a, st); forms::addc(b, st); }
79 => { forms::addc(a, st); forms::xorc(b, st); }
80 => { forms::addc(a, st); forms::mulc(b, st); }
81 => { forms::addc(a, st); forms::mulc2(b, st); }
82 => { forms::addc(a, st); forms::andx(b, st); }
83 => { forms::addc(a, st); forms::orx(b, st); }
84 => { forms::xorc(a, st); forms::add(b, st); }
85 => { forms::xorc(a, st); forms::sub(b, st); }
86 => { forms::xorc(a, st); forms::xor(b, st); }
87 => { forms::xorc(a, st); forms::mul(b, st); }
88 => { forms::xorc(a, st); forms::rot(b, st); }
89 => { forms::xorc(a, st); forms::xrot(b, st); }
90 => { forms::xorc(a, st); forms::addc(b, st); }
91 => { forms::xorc(a, st); forms::xorc(b, st); }
92 => { forms::xorc(a, st); forms::mulc(b, st); }
93 => { forms::xorc(a, st); forms::mulc2(b, st); }
94 => { forms::xorc(a, st); forms::andx(b, st); }
95 => { forms::xorc(a, st); forms::orx(b, st); }
96 => { forms::mulc(a, st); forms::add(b, st); }
97 => { forms::mulc(a, st); forms::sub(b, st); }
98 => { forms::mulc(a, st); forms::xor(b, st); }
99 => { forms::mulc(a, st); forms::mul(b, st); }
100 => { forms::mulc(a, st); forms::rot(b, st); }
101 => { forms::mulc(a, st); forms::xrot(b, st); }
102 => { forms::mulc(a, st); forms::addc(b, st); }
103 => { forms::mulc(a, st); forms::xorc(b, st); }
104 => { forms::mulc(a, st); forms::mulc(b, st); }
105 => { forms::mulc(a, st); forms::mulc2(b, st); }
106 => { forms::mulc(a, st); forms::andx(b, st); }
107 => { forms::mulc(a, st); forms::orx(b, st); }
108 => { forms::mulc2(a, st); forms::add(b, st); }
109 => { forms::mulc2(a, st); forms::sub(b, st); }
110 => { forms::mulc2(a, st); forms::xor(b, st); }
111 => { forms::mulc2(a, st); forms::mul(b, st); }
112 => { forms::mulc2(a, st); forms::rot(b, st); }
113 => { forms::mulc2(a, st); forms::xrot(b, st); }
114 => { forms::mulc2(a, st); forms::addc(b, st); }
115 => { forms::mulc2(a, st); forms::xorc(b, st); }
116 => { forms::mulc2(a, st); forms::mulc(b, st); }
117 => { forms::mulc2(a, st); forms::mulc2(b, st); }
118 => { forms::mulc2(a, st); forms::andx(b, st); }
119 => { forms::mulc2(a, st); forms::orx(b, st); }
120 => { forms::andx(a, st); forms::add(b, st); }
121 => { forms::andx(a, st); forms::sub(b, st); }
122 => { forms::andx(a, st); forms::xor(b, st); }
123 => { forms::andx(a, st); forms::mul(b, st); }
124 => { forms::andx(a, st); forms::rot(b, st); }
125 => { forms::andx(a, st); forms::xrot(b, st); }
126 => { forms::andx(a, st); forms::addc(b, st); }
127 => { forms::andx(a, st); forms::xorc(b, st); }
128 => { forms::andx(a, st); forms::mulc(b, st); }
129 => { forms::andx(a, st); forms::mulc2(b, st); }
130 => { forms::andx(a, st); forms::andx(b, st); }
131 => { forms::andx(a, st); forms::orx(b, st); }
132 => { forms::orx(a, st); forms::add(b, st); }
133 => { forms::orx(a, st); forms::sub(b, st); }
134 => { forms::orx(a, st); forms::xor(b, st); }
135 => { forms::orx(a, st); forms::mul(b, st); }
136 => { forms::orx(a, st); forms::rot(b, st); }
137 => { forms::orx(a, st); forms::xrot(b, st); }
138 => { forms::orx(a, st); forms::addc(b, st); }
139 => { forms::orx(a, st); forms::xorc(b, st); }
140 => { forms::orx(a, st); forms::mulc(b, st); }
141 => { forms::orx(a, st); forms::mulc2(b, st); }
142 => { forms::orx(a, st); forms::andx(b, st); }
143 => { forms::orx(a, st); forms::orx(b, st); }
_ => unreachable!(),
}
}
for ins in it.remainder() {
run_instr(ins, st);
}
}
/// The scalar reference: one instruction on one 16-word state, the text the kernels carry (`emit.rs`,
/// `derive_instr_text`) restated in Rust. The tests pin the SoA interpreter against it.
pub fn run_round_scalar(prog: &[DInstr], s: &mut [u32; DERIVE_REGS]) {
for ins in prog {
let d = ins.dst as usize;
let c = s[ins.src as usize];
match ins.op {
DOp::Add => s[d] = s[d].wrapping_add(c),
DOp::Sub => s[d] = s[d].wrapping_sub(c),
DOp::Xor => s[d] ^= c,
DOp::Mul => s[d] = s[d].wrapping_mul(c | 1),
DOp::Rot => s[d] = rotl(s[d], ins.rot as u32).wrapping_add(c),
DOp::XRot => s[d] = rotl(s[d] ^ c, ins.rot as u32),
DOp::AddC => s[d] = s[d].wrapping_add(c.wrapping_add(ins.imm)),
DOp::XorC => s[d] ^= c ^ ins.imm,
DOp::MulC => s[d] = (s[d] ^ c).wrapping_mul(ins.imm),
DOp::MulC2 => s[d] = s[d].wrapping_mul(ins.imm).wrapping_add(c),
DOp::AndX => s[d] ^= c & s[ins.src2 as usize],
DOp::OrX => s[d] = s[d].wrapping_add(c | s[ins.src2 as usize]),
}
}
}
/// The source text of one instruction in the C-family dialects (the same text in Metal, CUDA C and OpenCL C:
/// `s` is the 16-word state, `mh_rotl` the rotate of the memhard core).
pub fn instr_text(ins: &DInstr) -> String {
let (d, c, b) = (ins.dst, ins.src, ins.src2);
match ins.op {
DOp::Add => format!("s[{d}] += s[{c}];"),
DOp::Sub => format!("s[{d}] -= s[{c}];"),
DOp::Xor => format!("s[{d}] ^= s[{c}];"),
DOp::Mul => format!("s[{d}] *= (s[{c}] | 1u);"),
DOp::Rot => format!("s[{d}] = mh_rotl(s[{d}], {}u) + s[{c}];", ins.rot),
DOp::XRot => format!("s[{d}] = mh_rotl(s[{d}] ^ s[{c}], {}u);", ins.rot),
DOp::AddC => format!("s[{d}] += s[{c}] + {:#010x}u;", ins.imm),
DOp::XorC => format!("s[{d}] ^= s[{c}] ^ {:#010x}u;", ins.imm),
DOp::MulC => format!("s[{d}] = (s[{d}] ^ s[{c}]) * {:#010x}u;", ins.imm),
DOp::MulC2 => format!("s[{d}] = s[{d}] * {:#010x}u + s[{c}];", ins.imm),
DOp::AndX => format!("s[{d}] ^= (s[{c}] & s[{b}]);"),
DOp::OrX => format!("s[{d}] += (s[{c}] | s[{b}]);"),
}
}
/// One instruction as a line of program.json: `"add d=3 c=0"`, `"mulc d=5 c=3 imm=0x..."`.
pub fn instr_line(ins: &DInstr) -> String {
let mut s = format!("{} d={} c={}", ins.op.name(), ins.dst, ins.src);
if ins.op.has_third() {
s.push_str(&format!(" b={}", ins.src2));
}
if ins.op.has_rot() {
s.push_str(&format!(" k={}", ins.rot));
}
if ins.op.has_imm() {
s.push_str(&format!(" imm={:#010x}", ins.imm));
}
s
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn weights_sum_and_ops() {
assert_eq!(DOP_WEIGHTS.iter().map(|(_, w)| w).sum::<u64>(), 100);
for (i, (op, _)) in DOP_WEIGHTS.iter().enumerate() {
assert_eq!(*op as usize, i);
assert_eq!(DOp::from_u8(i as u8), Some(*op));
}
assert_eq!(DOp::for_roll(0), DOp::Add);
assert_eq!(DOp::for_roll(13), DOp::Add);
assert_eq!(DOp::for_roll(14), DOp::Sub);
assert_eq!(DOp::for_roll(99), DOp::OrX);
// the expected chip operations per instruction, 1.52 (1.62 on a GPU), put 736 x 9 over the x8 floors
let mean: f64 = DOP_WEIGHTS.iter().map(|(o, w)| o.chip_ops() as f64 * *w as f64 / 100.0).sum();
assert!((mean - 1.52).abs() < 1e-9, "{mean}");
let gpu_mean: f64 = DOP_WEIGHTS.iter().map(|(o, w)| o.gpu_ops() as f64 * *w as f64 / 100.0).sum();
assert!((gpu_mean - 1.62).abs() < 1e-9, "{gpu_mean}");
assert!(9.0 * DERIVE_LEN_X8 as f64 * mean > OPS_FLOOR_X8 as f64);
assert!(9.0 * DERIVE_LEN_X8 as f64 * gpu_mean > GPU_OPS_FLOOR_X8 as f64);
assert_eq!(DeriveProgram::floors(DERIVE_LEN_X8), (9_216, 10_368, 1_152));
assert_eq!(DeriveProgram::floors(368), (4_608, 5_184, 576));
let mul_share: f64 = DOP_WEIGHTS.iter().filter(|(o, _)| o.is_mul()).map(|(_, w)| *w as f64 / 100.0).sum();
assert!(9.0 * DERIVE_LEN_X8 as f64 * mul_share > MULS_FLOOR_X8 as f64);
}
#[test]
fn draw_is_structural_and_accepted() {
let mut rng = SplitMix64::new(0x1234_5678_9abc_def0);
let p = DeriveProgram::draw(&mut rng, DERIVE_LEN_X8);
assert_eq!(p.attempt, 0, "the first candidate of this seed passes");
assert_eq!(p.rounds.len(), DERIVE_PROGRAMS);
assert_eq!(p.instr_count(), 9 * DERIVE_LEN_X8 as u64);
assert!(p.check().is_ok());
assert!(p.chip_ops() >= OPS_FLOOR_X8 && p.gpu_ops() >= GPU_OPS_FLOOR_X8 && p.muls() >= MULS_FLOOR_X8);
assert!(p.gpu_ops() > p.chip_ops());
// every instruction consumes the newest result
for prog in &p.rounds {
let mut chain = 0u8;
for ins in prog {
assert_eq!(ins.src, chain);
assert_ne!(ins.dst, chain);
chain = ins.dst;
}
}
// four draws per instruction: the same program again from the same seed, and a different one one draw on
let mut rng2 = SplitMix64::new(0x1234_5678_9abc_def0);
assert_eq!(DeriveProgram::draw(&mut rng2, DERIVE_LEN_X8), p);
let mut rng3 = SplitMix64::new(0x1234_5678_9abc_def0);
rng3.next();
assert_ne!(DeriveProgram::draw(&mut rng3, DERIVE_LEN_X8), p);
}
#[test]
fn soa_matches_scalar_and_is_a_bijection() {
let mut rng = SplitMix64::new(7);
let p = DeriveProgram::draw(&mut rng, 64);
let mut st: SoaState = [[0u32; SOA_LANES]; DERIVE_REGS];
let mut scalars = [[0u32; DERIVE_REGS]; SOA_LANES];
let mut x = SplitMix64::new(99);
for k in 0..SOA_LANES {
for r in 0..DERIVE_REGS {
let v = x.next() as u32;
st[r][k] = v;
scalars[k][r] = v;
}
}
let before = scalars;
for prog in &p.rounds {
run_round(prog, &mut st);
for k in 0..SOA_LANES {
run_round_scalar(prog, &mut scalars[k]);
}
}
for k in 0..SOA_LANES {
for r in 0..DERIVE_REGS {
assert_eq!(st[r][k], scalars[k][r], "lane {k} reg {r}");
}
}
// distinct inputs stay distinct (a bijection on the state, spot-checked: 32 lanes, no collision)
for a in 0..SOA_LANES {
for b in a + 1..SOA_LANES {
assert_ne!(scalars[a], scalars[b]);
assert_ne!(before[a], before[b]);
}
}
}
#[test]
fn acceptance_rejects_degenerate_draws() {
let mut rng = SplitMix64::new(3);
let mut p = DeriveProgram::draw(&mut rng, 64);
// a register never written: make every write of round 2 go to the chain's neighbour
let mut q = p.clone();
for ins in q.rounds[2].iter_mut() {
ins.dst = if ins.src == 1 { 2 } else { 1 };
}
let mut chain = 0u8;
for ins in q.rounds[2].iter_mut() {
ins.src = chain;
ins.dst = if chain == 1 { 2 } else { 1 };
chain = ins.dst;
}
assert!(matches!(q.check(), Err(DeriveReject::RegisterNeverWritten { round: 2, .. })));
// all rotations equal
let mut q = p.clone();
for ins in q.rounds.iter_mut().flatten() {
if ins.op.has_rot() {
ins.rot = 5;
}
}
assert!(matches!(q.check(), Err(DeriveReject::RotationsDegenerate { distinct: 1 })));
// every op an add apart from the xor-rotates (so the rotations stay distinct): under the ops floor
for ins in p.rounds.iter_mut().flatten() {
if ins.op != DOp::XRot {
ins.op = DOp::Add;
ins.rot = 0;
ins.imm = 0;
ins.src2 = 0;
}
}
assert!(matches!(p.check(), Err(DeriveReject::OpsUnderFloor { .. })), "{:?}", p.check());
// no multiplies at all but the ops floor met: under the multiply floor
let mut q = DeriveProgram::draw(&mut SplitMix64::new(11), 64);
for ins in q.rounds.iter_mut().flatten() {
if ins.op.is_mul() {
ins.op = DOp::AddC;
}
}
assert!(matches!(q.check(), Err(DeriveReject::MulsUnderFloor { .. })), "{:?}", q.check());
}
#[test]
fn text_forms() {
let i = DInstr { op: DOp::MulC, dst: 5, src: 3, src2: 0, rot: 0, imm: 0x9e37_79b9 };
assert_eq!(instr_text(&i), "s[5] = (s[5] ^ s[3]) * 0x9e3779b9u;");
assert_eq!(instr_line(&i), "mulc d=5 c=3 imm=0x9e3779b9");
let i = DInstr { op: DOp::XRot, dst: 0, src: 15, src2: 0, rot: 17, imm: 0 };
assert_eq!(instr_text(&i), "s[0] = mh_rotl(s[0] ^ s[15], 17u);");
let i = DInstr { op: DOp::AndX, dst: 2, src: 9, src2: 14, rot: 0, imm: 0 };
assert_eq!(instr_text(&i), "s[2] ^= (s[9] & s[14]);");
}
}

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,43 @@
//! igneum-pow: the Igneum lottery hash, bit-exact with the Swift prototype in `proto-metal/main.swift`.
//!
//! The crate has five parts, each mirroring one section of the prototype:
//!
//! * [`seed`]: the 32-byte seed words from a string (FNV-1a 64, four salts) and the SplitMix64 stream.
//! * [`generator`]: the 64-instruction program drawn from a seed (version 2: 16 load slots, fresh sources).
//! * [`accept`]: the acceptance rule every candidate program must pass; a rejected candidate is replaced by the
//! next attempt of the same seed.
//! * [`memhard`]: the 256 MiB ChaCha12 cache and the 8-round dataset item derivation (`proto-metal/MEMHARD.md`).
//! * [`derive`]: the per-day item-derivation program of Counter ASIC 3.0 item 2 (a prototype behind a load class).
//! * [`verify`]: the 32-lane warp interpreter that computes the 64-bit hash on the CPU, deriving dataset
//! words on demand from the cache (or from the closed form, for the old packs).
//! * [`emit`]: the Metal, CUDA and OpenCL kernel text for a program, byte-identical to the Swift exporter.
//! * [`bind`]: the header binding (init words from the pre-PoW header hash and the nonce) and the 256-bit mapping.
//!
//! Nothing here depends on a crate outside the standard library. The integration points for the
//! rusty-kaspa fork (`docs/fork-map.md`) are [`verify::Epoch`], [`verify::Epoch::verify_block`] and
//! [`verify::Epoch::hash_warp`]; the node hands [`emit::Pack`] files to miners.
// The lane loops are written index style on purpose so they read like the kernels they mirror, and the
// SplitMix64 `next` keeps the Swift name.
#![allow(clippy::needless_range_loop, clippy::should_implement_trait, clippy::large_enum_variant)]
#![allow(clippy::manual_slice_size_calculation, clippy::too_many_arguments)]
pub mod accept;
pub mod bind;
pub mod blake2b;
pub mod derive;
pub mod emit;
pub mod generator;
pub mod memhard;
pub mod packcheck;
pub mod seed;
pub mod state;
pub mod verify;
pub use bind::{block_init_words, day_bytes, pow256_from_lane, target64_from_le256};
pub use accept::{check as accept_program, AcceptReport, Reject};
pub use generator::{generate, generate_from_seed_bytes, generate_from_seed_bytes_program_class, generate_from_seed_bytes_program_class_shadow, v4_class_at, v4_counted_ops, v4_rung_reps, v5_class_at, v5_rung_reps, Instr, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, GENERATOR_VERSION_V5, V3_CLASS, V4_CLASS, V4_SHADOW_INSTRS, V4_SHADOW_REPS, V5_CLASS, PROGRAM_SUBVERSION_V4};
pub use memhard::{cache_log2_words, dataset_log2_words, days_since_genesis, growth_doublings, Cache, MemhardCpu, MixParams, Shape};
pub use seed::{fnv1a64, seed_words, SplitMix64};
pub use state::{StateLeaves, StateStream};
pub use verify::{hash_warp, interpret_warp_init, verify_block, DatasetMode, DatasetSource, Epoch};

View file

@ -0,0 +1,502 @@
//! igneum-pow CLI.
//!
//! igneum-pow bench --seed <s> [--day <d>] [--closed-form] [--dataset-log2 28] [--warps 20]
//! igneum-pow export --seed <s> --out <dir> [--day <d>] [--closed-form] [--dataset-log2 28] [--epoch-hex <64 hex> --day-hex <hex>]
//! igneum-pow hash --seed <s> --nonce <n> [--day <d>] [--closed-form] [--dataset-log2 28]
//! igneum-pow hash-bound --seed <s> --prehash <64 hex> --nonce <u64> [--day <d>] [--closed-form] [--dataset-log2 28]
//! [--epoch-hex <64 hex> --day-hex <hex>] byte seeds instead of strings (Epoch::from_seed_bytes)
//! igneum-pow accept --seed <s> [--epoch-hex <64 hex>] every candidate of the seed with its verdict (spec 01 section 1.4.6)
//! igneum-pow show --seed <s> [--epoch-hex <64 hex>] the accepted program, one instruction per line
//!
//! Read-width experiment (5 October 2026, docs/plans/read-width.md): `--class v2|w4|w16|w64|w64x4|p4,p16,p64[xN]`
//! on every command selects the load class (default v2, the lottery hash). Nothing in a v2 run changes.
//!
//! Era layout (5 October 2026, docs/plans/era-layout.md): `--era igneum-era-test/<n>` (a test era seed: the 32 bytes
//! of seed_words_from_bytes of the string, era index n) or `--era <n>:<64 hex>` (the chain's 32-byte era seed E_n)
//! turns the chosen class into its era class; `--era-widths 4` (default: the read-width decision of 5 October 2026
//! keeps v2's 4-byte load) is the allowed width set the era draws from; `4,16,64` lets the era draw the width.
use igneum_pow::generator::{EraParams, GENERATOR_VERSION_V3};
use igneum_pow::emit::export_pack;
use igneum_pow::generator::{LoadClass, ProgramClass};
use igneum_pow::memhard::{Cache, Shape};
use igneum_pow::seed::day_key;
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch, DEFAULT_DATASET_LOG2};
use std::time::Instant;
struct Args {
cmd: String,
seed: String,
day: String,
out: Option<String>,
closed_form: bool,
dataset_log2: u32,
warps: usize,
nonce: u64,
/// `hash-bound --count N`: N consecutive nonces from --nonce, one epoch build (gate G2, 5 October 2026).
count: u64,
prehash: String,
epoch_hex: Option<String>,
day_hex: Option<String>,
/// Class v5: the window's state stream file (`--state <file>`, the IGSD1 format of `igneum_pow::state`), whose
/// leaves every item of the dataset is keyed by.
state: Option<String>,
class: LoadClass,
/// Days since genesis for the cache growth rule of a class with `growth` (0: the genesis cache).
days: u64,
/// The program class (Counter ASIC 2.0 seam): v2 (default), v3 (V3_CLASS, generator 3) or v4 (V4_CLASS, generator 4).
program_class: Option<ProgramClass>,
/// The shadow pass count of a class v4 program at a rung of the latency ladder (`--shadow-reps`, 0 = the class's
/// own 27; `docs/design/latency-ladder.md`); read under `--program-class v4` only.
shadow_reps: u16,
/// The era seed bytes a class v3 chain program records (`--era-hex`).
era_hex: Option<String>,
/// Era layout: `--era igneum-era-test/<n>` or `--era <n>:<64 hex>` composes the era class over `--class` with
/// generator 3 and the era bytes recorded (the measurement packs: v2's mixer under the era layout).
era: Option<(u64, Vec<u8>, String)>,
era_widths: Vec<u8>,
}
/// `igneum-era-test/<n>` or `<n>:<64 hex>` -> (index, 32 era bytes, label).
fn parse_era(s: &str) -> Option<(u64, Vec<u8>, String)> {
if let Some(n) = s.strip_prefix("igneum-era-test/") {
let index: u64 = n.parse().ok()?;
return Some((index, EraParams::test_era_bytes(s).to_vec(), s.to_string()));
}
let (n, hex) = s.split_once(':')?;
let index: u64 = n.parse().ok()?;
let bytes = igneum_pow::bind::unhex(hex)?;
if bytes.len() != 32 {
return None;
}
Some((index, bytes, format!("igneum-era/{index}/{hex}")))
}
/// "4,16,64" (bytes) -> ascending words.
fn parse_widths(s: &str) -> Option<Vec<u8>> {
let mut v: Vec<u8> = s
.split(',')
.map(|x| match x.trim() {
"4" => Some(1u8),
"16" => Some(4),
"64" => Some(16),
_ => None,
})
.collect::<Option<Vec<_>>>()?;
v.sort_unstable();
v.dedup();
if v.is_empty() {
return None;
}
Some(v)
}
fn usage() -> ! {
eprintln!(
"igneum-pow <bench|export|hash> --seed <string> [--day 2026-10-03] [--closed-form] [--dataset-log2 28]\n\
\x20 bench [--warps 20] fill the cache, then time the CPU verifier per 32-lane warp\n\
\x20 export --out <dir> write the program pack (kernel.cu, kernel.cl, program.metal, memhard.h, ...)\n\
\x20 hash --nonce <n> print the 64-bit hash of one nonce (pack form, init words = seed words)\n\
\x20 hash-bound --prehash <64 hex> --nonce <u64> [--count N] print the header-bound hash (bind.rs) of one 64-bit nonce (N consecutive: \"nonce hash\" lines)\n\
\x20 accept every candidate of the seed (or --epoch-hex) with its acceptance verdict\n\
\x20 show the accepted program, one instruction per line\n\
\x20 --class C load class: v2 (default), mx4, mx8 (class v3: mixer x8, cache growth), dr<len> (Counter ASIC 3.0 item 2: the per-day derivation program, dr736 = the x8-equivalent), w4, w16, w64, w64x4, p4,p16,p64[xN], <class>m<mult>[g]\n\
\x20 also: w4, w16, w64, w64x4, p4,p16,p64[xN], <class>m<mult>[g], <class>+sh<S>x<R> (latency-shadow block of S ALU instructions x R passes per iteration, Counter ASIC 3.0 item 8)\n\
\x20 --days N days since genesis for the cache growth rule of a class with it (default 0: the 2^26-word cache)\n\
\x20 --program-class v2|v3|v4|v5 the program class of the seam (v3 = generator 3 on V3_CLASS, v4 = generator 4 on V4_CLASS = mx8+sh256x27, v5 = generator 5 on V5_CLASS = mx8+sh256x27+state, the chain's own derivation; --era-hex records the era seed)\n\
\x20 --state <file> class v5 (or any --class ...+state): the window's state stream (IGSD1 file, igneum-day-stream --out), whose leaves key every item\n\
\x20 --shadow-reps N class v4 at a rung of the latency ladder: the shadow block's pass count (0 = the class's own 27; docs/design/latency-ladder.md), with --program-class v4\n\
\x20 --era E era layout over --class: igneum-era-test/<n> or <n>:<64 hex> (the 32-byte era seed E_n)\n\
\x20 --era-widths 4[,16,64] the width set the era draws from, in bytes (default 4: pinned; more lets the era draw it)"
);
std::process::exit(2)
}
fn parse() -> Args {
let mut a = Args {
cmd: String::new(),
seed: "igneum-genesis".into(),
day: "2026-10-03".into(),
out: None,
closed_form: false,
state: None,
dataset_log2: DEFAULT_DATASET_LOG2,
warps: 20,
nonce: 0,
count: 1,
prehash: "00".repeat(32),
epoch_hex: None,
day_hex: None,
class: LoadClass::V2,
days: 0,
program_class: None,
era_hex: None,
shadow_reps: 0,
era: None,
era_widths: vec![1],
};
let mut it = std::env::args().skip(1);
a.cmd = it.next().unwrap_or_else(|| usage());
while let Some(k) = it.next() {
let mut val = || it.next().unwrap_or_else(|| usage());
match k.as_str() {
"--seed" => a.seed = val(),
"--day" => a.day = val(),
"--out" => a.out = Some(val()),
"--closed-form" => a.closed_form = true,
"--dataset-log2" => a.dataset_log2 = val().parse().unwrap_or_else(|_| usage()),
"--warps" => a.warps = val().parse().unwrap_or_else(|_| usage()),
"--nonce" => a.nonce = val().parse().unwrap_or_else(|_| usage()),
"--count" => a.count = val().parse().unwrap_or_else(|_| usage()),
"--prehash" => a.prehash = val(),
"--epoch-hex" => a.epoch_hex = Some(val()),
"--day-hex" => a.day_hex = Some(val()),
"--class" => a.class = LoadClass::parse(&val()).unwrap_or_else(|| usage()),
"--days" => a.days = val().parse().unwrap_or_else(|_| usage()),
"--program-class" => a.program_class = Some(ProgramClass::parse(&val()).unwrap_or_else(|| usage())),
"--state" => a.state = Some(val()),
"--era-hex" => a.era_hex = Some(val()),
"--shadow-reps" => a.shadow_reps = val().parse().unwrap_or_else(|_| usage()),
"--era" => a.era = Some(parse_era(&val()).unwrap_or_else(|| usage())),
"--era-widths" => a.era_widths = parse_widths(&val()).unwrap_or_else(|| usage()),
_ => usage(),
}
}
if let Some((_, bytes, _)) = &a.era {
a.class = LoadClass::era(a.class, bytes, &a.era_widths);
}
a
}
/// An era program is a class v3 program: generator 3 and the era bytes recorded (what the chain's
/// `Epoch::from_chain_seeds` does); the pack then carries IGNEUM_PROGRAM_CLASS "v3" and IGNEUM_ERA_SEED_HEX.
fn stamp_era(e: &mut Epoch, a: &Args) {
if let Some((_, bytes, _)) = &a.era {
// generator 4 on V4_CLASS (class v4, Counter ASIC 3.0), 3 on every other era program
e.program.generator = igneum_pow::generator::era_generator_of(&e.program.class);
e.program.era_bytes = Some(bytes.clone());
}
}
fn main() {
let a = parse();
let mode = if a.closed_form { DatasetMode::ClosedForm } else { DatasetMode::MemoryHard };
match a.cmd.as_str() {
"bench" => bench(&a, mode),
"export" => export(&a, mode),
"accept" => accept(&a),
"show" => show(&a),
"hash" => {
let (e, _) = epoch_of(&a, mode);
println!("{:016x}", e.hash(a.nonce as u32));
}
"hash-bound" => {
let bytes = igneum_pow::bind::unhex(&a.prehash).unwrap_or_else(|| usage());
let prehash: [u8; 32] = bytes.as_slice().try_into().unwrap_or_else(|_| usage());
// --epoch-hex / --day-hex: the chain's byte seeds (Epoch::from_seed_bytes), as the worker protocol carries them
let (e, _) = epoch_of(&a, mode);
if a.count > 1 {
// gate G2 (5 October 2026, ported from branch ca2-era for the class v4 gate run): `--count N` prints
// "nonce hash" for N consecutive 64-bit nonces from --nonce, one epoch build, to re-hash a worker's
// found lines (the same prehash, target ff..ff)
for k in 0..a.count {
let n = a.nonce.wrapping_add(k);
println!("{n} {:016x}", e.hash_bound(&prehash, n));
}
return;
}
let init = igneum_pow::bind::block_init_words(&prehash, a.nonce);
println!("init words {}", init.iter().map(|w| format!("{w:08x}")).collect::<Vec<_>>().join(" "));
println!("{:016x}", e.hash_bound(&prehash, a.nonce));
}
_ => usage(),
}
}
/// The epoch every command works on, and the day label for packs. `--epoch-hex`/`--day-hex` give the chain's byte
/// seeds (the day label then names the day bytes); else the string seed and day. `--program-class v3` draws the
/// program through the seam (generator 3 on `V3_CLASS`, the era bytes of `--era-hex` recorded) and sizes the
/// dataset for `--days` through `Epoch::chain_dataset_day`; `--class` is ignored under a program class (the class
/// names the load class). Closed-form mode is only for string seeds under the default class.
fn epoch_of(a: &Args, mode: DatasetMode) -> (Epoch, String) {
let (mut e, label) = epoch_of_class(a, mode);
stamp_era(&mut e, a);
// class v5: the leaves of --state, built for the dataset's size; a state class without --state is refused here
// rather than at the first derivation
if e.program.class.state {
let Some(path) = &a.state else {
eprintln!("class {} keys every item by the window's state: give --state <stream file> (igneum-day-stream --out)", e.program.class.name());
std::process::exit(2);
};
let stream = igneum_pow::state::StateStream::read_file(std::path::Path::new(path)).unwrap_or_else(|err| {
eprintln!("{err}");
std::process::exit(2)
});
let leaves = igneum_pow::state::StateLeaves::from_stream(&stream, e.dataset.log2_words);
eprintln!(
"state stream {}: chain block {} {}, root {}, {} records, {} leaves{}",
path,
stream.number,
igneum_pow::emit::hex_bytes(&stream.block),
igneum_pow::emit::hex_bytes(&stream.root),
stream.records.len(),
leaves.n(),
if leaves.sampled { " (sampled)" } else { "" }
);
e.dataset = e.dataset.with_leaves(std::sync::Arc::new(leaves));
} else if a.state.is_some() {
eprintln!("--state given for a class without state leaves ({}); use --program-class v5 or --class <class>+state", e.program.class.name());
std::process::exit(2);
}
(e, label)
}
fn epoch_of_class(a: &Args, mode: DatasetMode) -> (Epoch, String) {
let era = a.era_hex.as_ref().map(|h| igneum_pow::bind::unhex(h).unwrap_or_else(|| usage()));
match (&a.epoch_hex, &a.day_hex) {
(Some(eh), Some(dh)) => {
let eb = igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage());
let db = igneum_pow::bind::unhex(dh).unwrap_or_else(|| usage());
let label = format!("igneum-epoch/{eh}/day/{dh}");
let e = match a.program_class {
Some(pc) => Epoch {
program: Epoch::chain_program_shadow(&eb, era.as_deref(), pc, a.shadow_reps, &label),
dataset: Epoch::chain_dataset_day(&db, pc, a.days, a.dataset_log2),
},
None => Epoch::from_seed_bytes_day(&eb, &db, &label, a.class, a.days, a.dataset_log2),
};
(e, format!("bytes:{dh}"))
}
_ => {
let e = match a.program_class {
Some(pc) => {
let program = igneum_pow::generator::generate_from_seed_bytes_program_class_shadow(&a.seed, a.seed.as_bytes(), pc, era.as_deref(), a.shadow_reps);
let lc = pc.load_class();
let shape = Shape::for_class_day(&lc, a.days);
let log2 = if lc.growth { igneum_pow::memhard::dataset_log2_words(a.dataset_log2, a.days) } else { a.dataset_log2 };
Epoch { program, dataset: DatasetSource::new_shape(&a.day, mode, log2, shape) }
}
None => Epoch::new_class_day(&a.seed, &a.day, mode, a.dataset_log2, a.class, a.days),
};
(e, a.day.clone())
}
}
}
fn bench(a: &Args, mode: DatasetMode) {
println!(
"igneum-pow bench: seed \"{}\", day \"{}\", dataset 2^{} words ({})",
a.seed,
a.day,
a.dataset_log2,
mode.name()
);
let shape = Shape::for_class_day(&a.program_class.map(|pc| pc.load_class()).unwrap_or(a.class), a.days);
if mode == DatasetMode::MemoryHard {
// Time the cache fill on its own first (one core), then build the epoch (which fills it again).
let t0 = Instant::now();
let c = Cache::fill_log2(day_key(&a.day), shape.cache_log2_words);
let fill_ms = t0.elapsed().as_secs_f64() * 1e3;
println!(
"cache: fill {fill_ms:.1} ms on one core (2^{} words, {} MiB, {} chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}",
shape.cache_log2_words,
shape.cache_words() * 4 / (1 << 20),
c.segments(),
c.fnv1a64()
);
drop(c);
}
if let Some(h) = a.class.hot {
// the hot table of the epoch on its own first (one core), then the epoch (which fills it again)
let t0 = Instant::now();
let t = igneum_pow::memhard::HotTable::for_seed_bytes(a.seed.as_bytes(), h.mb as u32);
let fill_ms = t0.elapsed().as_secs_f64() * 1e3;
println!("hot table: {} MiB filled in {fill_ms:.1} ms on one core ({} chains of 64 ChaCha12 blocks), FNV-1a 64 {:016x}", h.mb, igneum_pow::memhard::hot_segments(h.mb as u32), t.fnv1a64());
}
let t0 = Instant::now();
let (e, _) = epoch_of(a, mode);
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
println!(
"program: class {}, {} loads/hash, {} bytes/hash, widths (1,4,16 words) {:?}, {} items/warp, mixer x{} ({} mixers/item), cache 2^{} words, op mix {}; epoch built in {build_ms:.1} ms",
e.program.class.name(),
e.program.loads_per_hash(),
e.program.bytes_per_hash(),
e.program.width_counts(),
e.program.items_per_warp(),
shape.mixer_mult,
shape.mixers_per_item(),
shape.cache_log2_words,
e.program.op_mix()
);
if let Some(dp) = e.dataset.memhard().and_then(|m| m.params.derive.as_ref()) {
// Counter ASIC 3.0 item 2: the day's derivation program
println!(
"derivation program: {} instructions per round program, {} per item ({} GPU ops, {} chip ops, {} multiplies per item), attempt {}, fingerprint {:016x}, op mix {}",
dp.len,
dp.instr_count(),
dp.gpu_ops(),
dp.chip_ops(),
dp.muls(),
dp.attempt,
dp.fingerprint(),
dp.op_mix()
);
}
if e.program.has_shadow() {
println!(
"shadow: {} instructions x {} passes per iteration, {} shadow instructions per hash, op mix {}",
e.program.shadow.len(),
e.program.shadow_reps(),
e.program.shadow_instrs_per_hash(),
e.program.shadow_op_mix()
);
}
let bases = [0u32, 4096, 1_000_000];
for &b in &bases {
let t = Instant::now();
let r = e.interpret_warp(b);
let ms = t.elapsed().as_secs_f64() * 1e3;
println!(
"warp base {b}: single cold run {ms:.3} ms, {} items derived, lane0 {:016x} lane31 {:016x}",
r.items_derived, r.hashes[0], r.hashes[31]
);
}
let n = a.warps.max(1);
let t = Instant::now();
let mut sink = 0u64;
for i in 0..n {
let w = e.hash_warp((i as u32) * 32 + 65536);
sink ^= w[0];
}
let avg = t.elapsed().as_secs_f64() * 1e3 / n as f64;
println!("CPU verify: {avg:.3} ms per 32-lane warp, avg of {n} (checksum {sink:016x})");
}
fn export(a: &Args, mode: DatasetMode) {
let out = a.out.clone().unwrap_or_else(|| usage());
let t0 = Instant::now();
// --epoch-hex / --day-hex: the chain's byte seeds; the day label then names the day bytes
let (e, day_label) = epoch_of(a, mode);
let build_ms = t0.elapsed().as_secs_f64() * 1e3;
println!("igneum-pow export {out}");
println!(
"seed \"{}\", day \"{}\", dataset 2^{} words ({}), generator v{} attempt {} program id {:016x}, loads/hash {}; epoch built in {build_ms:.1} ms",
e.program.seed_string,
day_label,
e.dataset.log2_words,
e.dataset.mode().name(),
e.program.generator,
e.program.attempt,
e.program.program_id(),
e.program.loads_per_hash()
);
println!("op mix: {}; class {}, {} bytes/hash, widths (1,4,16 words) {:?}", e.program.op_mix(), e.program.class.name(), e.program.bytes_per_hash(), e.program.width_counts());
let source = format!("igneum-pow (Rust) CPU interpreter, generator v{}, {} dataset", e.program.generator, e.dataset.mode().name());
let pack = export_pack(&e, &day_label, &source);
let dir = std::path::Path::new(&out);
if let Err(err) = pack.write_to(dir) {
eprintln!("FAIL: write error {err}");
std::process::exit(1);
}
for (name, text) in &pack.files {
println!("wrote {}/{name} ({} bytes)", dir.display(), text.len());
}
for (i, b) in pack.bases.iter().enumerate() {
println!("vector warp base {b}: lane0 {:016x} lane31 {:016x}", pack.outs[i][0], pack.outs[i][31]);
}
if e.dataset.mode() == DatasetMode::MemoryHard {
println!("cache FNV-1a 64 {:016x}", pack.vectors.cache_fnv);
}
println!("OVERALL: PASS (pack written)");
}
fn seed_bytes_of(a: &Args) -> (String, Vec<u8>) {
match &a.epoch_hex {
Some(eh) => (format!("igneum-epoch/{eh}"), igneum_pow::bind::unhex(eh).unwrap_or_else(|| usage())),
None => (a.seed.clone(), a.seed.as_bytes().to_vec()),
}
}
fn accept(a: &Args) {
let (label, bytes) = seed_bytes_of(a);
let t0 = Instant::now();
let tries = igneum_pow::generator::attempts_class(&label, &bytes, a.class);
let ms = t0.elapsed().as_secs_f64() * 1e3;
for (p, verdict) in &tries {
match verdict {
Ok(()) => {
let r = igneum_pow::accept::check(p).unwrap();
println!(
"attempt {}: ACCEPTED program id {:016x}, op mix {}, distinct {:.3} per hash, saturated {}, bias max {}",
p.attempt,
p.program_id(),
p.op_mix(),
r.distinct_mean(),
r.saturated,
r.bias_max
);
}
Err(r) => println!("attempt {}: rejected, {r}", p.attempt),
}
}
println!("{} candidates in {ms:.2} ms", tries.len());
}
fn show(a: &Args) {
let (label, bytes) = seed_bytes_of(a);
// --program-class: the chain's own derivation (the era from --era-hex), as the node and the miner draw it; else
// the load class of --class, the era of --era composed over it
let era_hex_bytes = a.era_hex.as_deref().map(|h| igneum_pow::bind::unhex(h).unwrap_or_else(|| usage()));
let mut p = match a.program_class {
Some(pc) => igneum_pow::generator::generate_from_seed_bytes_program_class_shadow(&label, &bytes, pc, era_hex_bytes.as_deref(), a.shadow_reps),
None => igneum_pow::generator::generate_from_seed_bytes_class(&label, &bytes, a.class),
};
if let (None, Some((_, eb, _))) = (a.program_class, &a.era) {
p.generator = igneum_pow::generator::era_generator_of(&p.class);
p.era_bytes = Some(eb.clone());
}
println!(
"seed \"{}\" generator v{} class {} attempt {} program id {:016x} seed words {}",
p.seed_string,
p.generator,
p.class.name(),
p.attempt,
p.program_id(),
p.seed.iter().map(|w| format!("{w:08x}")).collect::<Vec<_>>().join(" ")
);
println!("op mix {} loads/hash {} bytes/hash {}", p.op_mix(), p.loads_per_hash(), p.bytes_per_hash());
if let Some(e) = p.class.era {
println!(
"era {} ({}): width {} B, stride mul {:#010x} rot {}, interleave {:?}, windows (site:shrink:offset) {}",
e.label(),
a.era.as_ref().map(|x| x.2.as_str()).unwrap_or("?"),
e.width_words as u32 * 4,
e.stride_mul,
e.stride_rot,
e.pos,
p.instrs
.iter()
.enumerate()
.filter(|(_, i)| i.op == igneum_pow::generator::Op::Load)
.map(|(k, i)| format!("{k}:{}:{}", i.win, i.off))
.collect::<Vec<_>>()
.join(" ")
);
}
for (k, i) in p.instrs.iter().enumerate() {
println!(
"{k:2}: {:5} dst={} src={} src2={} imm={:#010x} imm2={:#010x} rot={} bit={} mask={}{}",
i.op.name(),
i.dst,
i.src,
i.src2,
i.imm,
i.imm2,
i.rot,
i.bit,
i.mask,
if i.op == igneum_pow::generator::Op::Load && i.width > 1 { format!(" width={}B", i.width as u32 * 4) } else { String::new() }
);
}
}

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,464 @@
//! A program pack on disk, read the way the one-click workers read it (`proto-cuda/nvrtc/packfile.h`, `pf_load`),
//! and checked against the seeds the node is on.
//!
//! The rule (5 October 2026, the epoch 34 incident on both PCs): a pack's `IGNEUM_SEEDW_INIT` is the seed words of
//! the program's ATTEMPT, `attempt_words(epoch_seed, IGNEUM_PROGRAM_ATTEMPT)`, not the words of the bare seed. The
//! generator retries a rejected candidate with `seed || k_le32` (spec 01 section 1.4.6), so from attempt 1 on the
//! bare-seed words and the pack's words differ. The workers derived the expected words from the bare seed, refused
//! every pack of a retried program ("the epoch seed bytes do not give the pack's IGNEUM_SEEDW_INIT") and the miner
//! and the app restarted them forever. Epoch 34 (seed `009858237e11...`) was the first live epoch whose program is
//! a later attempt. This module is the one place that rule is written in Rust; the miner checks every pack it
//! writes with it before a worker sees the pack, and the tests pin the attempt vectors the C side also pins.
use crate::generator::{attempt_words, ProgramClass};
use crate::seed::seed_words_from_bytes;
use std::fmt;
use std::path::Path;
/// What a pack says about itself.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct PackIdentity {
pub epoch_hex: String,
pub day_hex: String,
pub attempt: u32,
pub seedw: [u32; 8],
pub keyw: [u32; 8],
/// `IGNEUM_GENERATOR` (2, 3 or 4; a pack without the line is generator 1, which no worker runs).
pub generator: u32,
/// The program class the generator version names (Counter ASIC 2.0).
pub class: ProgramClass,
/// `IGNEUM_ERA_SEED_HEX` when the pack carries one (class v3 chain packs).
pub era_hex: Option<String>,
/// `IGNEUM_STATE_ROOT_HEX` of a class v5 pack (the window's state root the leaves derive from).
pub state_root_hex: Option<String>,
}
/// Why a pack is not the one a worker should mine with. `Display` is the plain-words line the logs carry.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum PackFault {
/// program.h or seeds.txt is missing or does not parse.
Unreadable(String),
/// A well-formed pack for other seeds than the node's: the pack is stale (or the node moved on).
OutOfDate { pack_epoch: String, pack_day: String, want_epoch: String, want_day: String },
/// The files of one pack contradict each other (seeds.txt against program.h, or the init words against the
/// seeds and the attempt): a half-written or hand-edited pack, or a worker and an exporter on different rules.
Disagree(String),
/// The pack is of another program class than the one the chain is on (spec 01 section 1.4.5: an implementation
/// refuses a pack whose generator version is not its own), or its era seed is not the era the job names.
WrongClass(String),
}
impl fmt::Display for PackFault {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
match self {
PackFault::Unreadable(w) => write!(f, "program pack unreadable: {w}"),
PackFault::OutOfDate { pack_epoch, pack_day, want_epoch, want_day } => write!(
f,
"program pack out of date: the pack is for epoch {} day {}, the node is on epoch {} day {}",
short(pack_epoch),
day_label(pack_day),
short(want_epoch),
day_label(want_day)
),
PackFault::Disagree(w) => write!(f, "program pack and its seeds disagree: {w}"),
PackFault::WrongClass(w) => write!(f, "program pack of the wrong class: {w}"),
}
}
}
impl std::error::Error for PackFault {}
fn short(hex: &str) -> &str {
if hex.len() >= 16 {
&hex[..16]
} else {
hex
}
}
/// The day bytes are `igneum-day/` followed by the little-endian day index (`bind::day_bytes`); print the index
/// when the hex has that shape, else the hex.
fn day_label(hex: &str) -> String {
const PREFIX: &str = "69676e65756d2d6461792f"; // "igneum-day/"
if let Some(rest) = hex.strip_prefix(PREFIX) {
if let Some(bytes) = unhex(rest) {
let mut v = 0u64;
for (i, b) in bytes.iter().enumerate().take(8) {
v |= (*b as u64) << (8 * i);
}
return v.to_string();
}
}
hex.to_string()
}
pub fn hex(bytes: &[u8]) -> String {
bytes.iter().map(|b| format!("{b:02x}")).collect()
}
fn unhex(s: &str) -> Option<Vec<u8>> {
if s.len() % 2 != 0 {
return None;
}
(0..s.len()).step_by(2).map(|i| u8::from_str_radix(&s[i..i + 2], 16).ok()).collect()
}
/// `#define NAME <rest of line>` in a header; the value as text, trimmed, with a trailing `//` comment removed.
fn define(text: &str, name: &str) -> Option<String> {
for line in text.lines() {
let t = line.trim_start();
let Some(rest) = t.strip_prefix("#define ") else { continue };
let rest = rest.trim_start();
let Some(after) = rest.strip_prefix(name) else { continue };
if !after.starts_with(|c: char| c.is_whitespace()) {
continue;
}
let v = after.trim();
let v = v.split("//").next().unwrap_or("").trim();
return Some(v.to_string());
}
None
}
fn define_str(text: &str, name: &str) -> Option<String> {
let v = define(text, name)?;
let v = v.strip_prefix('"')?.strip_suffix('"')?;
Some(v.to_string())
}
fn define_u32(text: &str, name: &str) -> Option<u32> {
let v = define(text, name)?;
let v = v.trim_end_matches('u');
if let Some(h) = v.strip_prefix("0x") {
u32::from_str_radix(h, 16).ok()
} else {
v.parse().ok()
}
}
fn define_words(text: &str, name: &str) -> Option<[u32; 8]> {
let v = define(text, name)?;
let inner = v.trim().strip_prefix('{')?.strip_suffix('}')?;
let mut out = [0u32; 8];
let mut n = 0;
for part in inner.split(',') {
let p = part.trim().trim_end_matches('u');
if p.is_empty() {
continue;
}
if n >= 8 {
return None;
}
out[n] = if let Some(h) = p.strip_prefix("0x") { u32::from_str_radix(h, 16).ok()? } else { p.parse().ok()? };
n += 1;
}
(n == 8).then_some(out)
}
/// One `key value` line of seeds.txt.
fn seeds_line(text: &str, key: &str) -> Option<String> {
text.lines().find_map(|l| l.strip_prefix(key).and_then(|r| r.strip_prefix(' ')).map(|v| v.trim().to_string()))
}
/// Checks the texts of a pack (program.h, and seeds.txt when it exists) against the seeds a worker will be asked
/// to mine with. Pure: the miner and the tests call it with file contents.
pub fn verify_pack_texts(program_h: &str, seeds_txt: Option<&str>, want_epoch: &[u8], want_day: &[u8]) -> Result<PackIdentity, PackFault> {
verify_pack_texts_chain(program_h, seeds_txt, want_epoch, want_day, None, None)
}
/// [`verify_pack_texts`] that also demands a program class and, for class v3 and v4, the era seed the chain is on
/// (Counter ASIC 2.0, 5 October 2026; class v4 Counter ASIC 3.0, 6 October 2026). `want_class` `None` accepts any
/// class; `want_era` `None` skips the era. A pack whose `IGNEUM_GENERATOR` is not 2, 3 or 4 is refused whatever is wanted.
pub fn verify_pack_texts_chain(
program_h: &str,
seeds_txt: Option<&str>,
want_epoch: &[u8],
want_day: &[u8],
want_class: Option<ProgramClass>,
want_era: Option<&[u8]>,
) -> Result<PackIdentity, PackFault> {
let generator = define_u32(program_h, "IGNEUM_GENERATOR").unwrap_or(1);
let Some(class) = ProgramClass::from_generator(generator) else {
return Err(PackFault::WrongClass(format!("IGNEUM_GENERATOR {generator} is not a generator version this software runs (2, 3 or 4)")));
};
// IGNEUM_PROGRAM_CLASS, when present, must name the class the generator version names
if let Some(named) = define_str(program_h, "IGNEUM_PROGRAM_CLASS") {
if ProgramClass::parse(&named) != Some(class) {
return Err(PackFault::Disagree(format!("IGNEUM_PROGRAM_CLASS {named:?} does not match IGNEUM_GENERATOR {generator}")));
}
}
let era_hex = define_str(program_h, "IGNEUM_ERA_SEED_HEX").map(|h| h.to_ascii_lowercase());
// Counter ASIC 3.0 (6 October 2026): the shadow block is the mark of class v4, so a generator 4 pack carries
// IGNEUM_SHADOW_INSTRS and a generator 3 pack does not; a pack exported as class v4 but stamped generator 3 (the
// seven gate packs of 6 October, which carried the v3 control's program id because `program_id(3, ..)` is
// class-independent) is refused here instead of mining as class v3 under the wrong id, and a generator 4 pack
// without the block is no v4 pack. A generator 2 pack with a shadow (the measurement ladder of
// proto-cuda/packs-ca3-shadow) carries a class-bearing id and stays loadable.
let shadow = define_u32(program_h, "IGNEUM_SHADOW_INSTRS").unwrap_or(0);
// Class v5 (docs/design/class-v5-stored-state.md): the state lines are the mark of class v5, so a generator 5 pack
// carries IGNEUM_STATE_ROOT_HEX (and the shadow block of v4) and no other generator does.
let state_root_hex = define_str(program_h, "IGNEUM_STATE_ROOT_HEX");
match (class, state_root_hex.is_some()) {
(ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_STATE_ROOT_HEX: not a class v5 pack".into())),
(ProgramClass::V5, true) => {}
(_, true) => return Err(PackFault::Disagree(format!("IGNEUM_GENERATOR {generator} with class v5 state lines: a program over state leaves is generator 5 (export the pack as class v5)"))),
_ => {}
}
match (class, shadow > 0) {
(ProgramClass::V4, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 4 (class v4) without IGNEUM_SHADOW_INSTRS: not a class v4 pack".into())),
(ProgramClass::V5, false) => return Err(PackFault::Disagree("IGNEUM_GENERATOR 5 (class v5) without IGNEUM_SHADOW_INSTRS: not a class v5 pack".into())),
// the class v4 stream sub-version (AP-F8-1 amendment, 7 October 2026): a generator 4 pack from before the
// load-source rule carries no IGNEUM_PROGRAM_SUBVERSION and its program id is another stream's; refused
(ProgramClass::V4, true) if define_u32(program_h, "IGNEUM_PROGRAM_SUBVERSION") != Some(u32::from(crate::generator::PROGRAM_SUBVERSION_V4)) => {
return Err(PackFault::Disagree(format!(
"IGNEUM_GENERATOR 4 (class v4) with IGNEUM_PROGRAM_SUBVERSION {}: this software runs sub-version {} (a class v4 pack from before the load-source rule, or after another amendment)",
define_u32(program_h, "IGNEUM_PROGRAM_SUBVERSION").map(|v| v.to_string()).unwrap_or_else(|| "absent".into()),
crate::generator::PROGRAM_SUBVERSION_V4
)))
}
(ProgramClass::V3, true) => {
return Err(PackFault::Disagree(format!("IGNEUM_GENERATOR 3 (class v3) with a shadow block (IGNEUM_SHADOW_INSTRS {shadow}): a class v4 program is generator 4 (export the pack as class v4)")))
}
_ => {}
}
if let Some(want) = want_class {
if want != class {
return Err(PackFault::WrongClass(format!("the pack is program class {} (generator {generator}), the chain is on class {}", class.name(), want.name())));
}
}
if let (Some(want), true) = (want_era, class.has_era()) {
let want_hex = hex(want);
match &era_hex {
Some(h) if *h == want_hex => {}
Some(h) => return Err(PackFault::WrongClass(format!("the pack's era seed {} is not the era seed {} the job names", short(h), short(&want_hex)))),
None => return Err(PackFault::WrongClass(format!("a class {} pack without IGNEUM_ERA_SEED_HEX; the job names an era seed", class.name()))),
}
}
let seedw = define_words(program_h, "IGNEUM_SEEDW_INIT").ok_or_else(|| PackFault::Unreadable("program.h has no IGNEUM_SEEDW_INIT with 8 words".into()))?;
let keyw = define_words(program_h, "IGNEUM_KEY_INIT").ok_or_else(|| PackFault::Unreadable("program.h has no IGNEUM_KEY_INIT with 8 words".into()))?;
let attempt = define_u32(program_h, "IGNEUM_PROGRAM_ATTEMPT").unwrap_or(0);
let mut epoch_hex = define_str(program_h, "IGNEUM_SEED_BYTES_HEX").unwrap_or_default();
let mut day_hex = define_str(program_h, "IGNEUM_DAY_BYTES_HEX").unwrap_or_default();
if let Some(s) = seeds_txt {
let e = seeds_line(s, "epoch_seed_hex").ok_or_else(|| PackFault::Unreadable("seeds.txt has no epoch_seed_hex line".into()))?;
let d = seeds_line(s, "day_seed_hex").ok_or_else(|| PackFault::Unreadable("seeds.txt has no day_seed_hex line".into()))?;
if !epoch_hex.is_empty() && !epoch_hex.eq_ignore_ascii_case(&e) {
return Err(PackFault::Disagree(format!("seeds.txt names epoch {} but program.h was generated for epoch {} (a pack half rewritten?)", short(&e), short(&epoch_hex))));
}
if !day_hex.is_empty() && !day_hex.eq_ignore_ascii_case(&d) {
return Err(PackFault::Disagree(format!("seeds.txt names day {} but program.h was generated for day {}", day_label(&d), day_label(&day_hex))));
}
epoch_hex = e.to_ascii_lowercase();
day_hex = d.to_ascii_lowercase();
}
if epoch_hex.is_empty() || day_hex.is_empty() {
return Err(PackFault::Unreadable("no seeds: neither seeds.txt nor IGNEUM_SEED_BYTES_HEX / IGNEUM_DAY_BYTES_HEX in program.h".into()));
}
let epoch_bytes = unhex(&epoch_hex).filter(|b| b.len() == 32).ok_or_else(|| PackFault::Unreadable("the epoch seed is not 32 bytes of hex".into()))?;
let day_bytes = unhex(&day_hex).ok_or_else(|| PackFault::Unreadable("the day seed hex is malformed".into()))?;
// The pack's own consistency first: a pack that contradicts itself is never "out of date", it is broken
let want_w = attempt_words(&epoch_bytes, attempt);
if want_w != seedw {
return Err(PackFault::Disagree(format!(
"IGNEUM_SEEDW_INIT is not attempt {attempt} of the epoch seed {} (the words of attempt {attempt} are {:08x} {:08x} ..., the pack has {:08x} {:08x} ...)",
short(&epoch_hex),
want_w[0],
want_w[1],
seedw[0],
seedw[1]
)));
}
let want_k = seed_words_from_bytes(&day_bytes);
if want_k != keyw {
return Err(PackFault::Disagree(format!("IGNEUM_KEY_INIT is not the key of the day seed {} ", day_label(&day_hex))));
}
let want_epoch_hex = hex(want_epoch);
let want_day_hex = hex(want_day);
if epoch_hex != want_epoch_hex || day_hex != want_day_hex {
return Err(PackFault::OutOfDate { pack_epoch: epoch_hex, pack_day: day_hex, want_epoch: want_epoch_hex, want_day: want_day_hex });
}
Ok(PackIdentity { epoch_hex, day_hex, attempt, seedw, keyw, generator, class, era_hex, state_root_hex })
}
/// [`verify_pack_texts`] over a pack directory.
pub fn verify_pack_dir(dir: &Path, want_epoch: &[u8], want_day: &[u8]) -> Result<PackIdentity, PackFault> {
verify_pack_dir_chain(dir, want_epoch, want_day, None, None)
}
/// [`verify_pack_texts_chain`] over a pack directory.
pub fn verify_pack_dir_chain(dir: &Path, want_epoch: &[u8], want_day: &[u8], want_class: Option<ProgramClass>, want_era: Option<&[u8]>) -> Result<PackIdentity, PackFault> {
let program_h = std::fs::read_to_string(dir.join("program.h")).map_err(|e| PackFault::Unreadable(format!("cannot read {}/program.h: {e}", dir.display())))?;
let seeds = std::fs::read_to_string(dir.join("seeds.txt")).ok();
verify_pack_texts_chain(&program_h, seeds.as_deref(), want_epoch, want_day, want_class, want_era)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::emit::program_header;
use crate::generator::generate_from_seed_bytes;
use crate::verify::Epoch;
// The two live devnet epochs of 5 October 2026 (epoch 33 mined, epoch 34 refused by the one-click workers)
const EPOCH_33: &str = "bed7ab62cbece66cf791485336d81d90fa1452ffed28ecd8a7416960ef64164c";
const EPOCH_34: &str = "009858237e118f69abc8d096e9b1af21c24539eaecdfd1b896588825660a69ec";
// "igneum-day/" || le64(20731)
const DAY_20731: &str = "69676e65756d2d6461792ffb50000000000000";
fn bytes(h: &str) -> Vec<u8> {
unhex(h).unwrap()
}
/// The attempt vectors the C side pins too (proto-cuda/nvrtc/emu/packfile-test.c): a change to either
/// derivation fails on one side first.
#[test]
fn attempt_words_vectors_shared_with_the_workers() {
let e = bytes(EPOCH_34);
assert_eq!(attempt_words(&e, 0), [0x06af2a61, 0x4d67274e, 0x4ebda738, 0xad1dea73, 0x6233cd8c, 0x50371601, 0x39d0b873, 0x6af024a2]);
assert_eq!(attempt_words(&e, 1), [0x0dcff56b, 0x6b1beb0d, 0x234dc70c, 0xe4016fa9, 0x72397152, 0xb558aa79, 0x3ffb3299, 0x72b9962e]);
// epoch 33's bare words, as the CUDA worker printed them on 5 October ("seed words be5983a6 f750dab7 ...")
assert_eq!(attempt_words(&bytes(EPOCH_33), 0)[..2], [0xbe5983a6, 0xf750dab7]);
}
/// The incident: epoch 34's program is a later attempt, epoch 33's is the bare seed. A worker that derives the
/// words from the bare seed accepts 33 and refuses 34.
#[test]
fn epoch_34_program_is_a_later_attempt() {
let p34 = generate_from_seed_bytes("epoch 34", &bytes(EPOCH_34));
assert!(p34.attempt >= 1, "epoch 34 must be a retried program for the incident to reproduce; attempt {}", p34.attempt);
assert_eq!(p34.seed, attempt_words(&bytes(EPOCH_34), p34.attempt));
assert_ne!(p34.seed, attempt_words(&bytes(EPOCH_34), 0));
let p33 = generate_from_seed_bytes("epoch 33", &bytes(EPOCH_33));
assert_eq!(p33.attempt, 0);
}
fn pack_texts(epoch_hex: &str, day_hex: &str) -> (String, String, u32) {
let (e, d) = (bytes(epoch_hex), bytes(day_hex));
let epoch = Epoch::from_seed_bytes(&e, &d, "test");
let h = program_header(&epoch.program, "test day", &epoch.dataset);
let s = format!("epoch_seed_hex {epoch_hex}\nday_seed_hex {day_hex}\nday_index 20731\n");
(h, s, epoch.program.attempt)
}
/// Known-good: the pack of a retried program verifies against its own seeds, with its attempt.
#[test]
fn known_good_pack_of_a_later_attempt_verifies() {
let (h, s, attempt) = pack_texts(EPOCH_34, DAY_20731);
assert!(attempt >= 1);
let id = verify_pack_texts(&h, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).expect("the pack verifies");
assert_eq!(id.attempt, attempt);
assert_eq!(id.epoch_hex, EPOCH_34);
assert_eq!(id.seedw, attempt_words(&bytes(EPOCH_34), attempt));
// without seeds.txt program.h's own bytes carry the pack
assert!(verify_pack_texts(&h, None, &bytes(EPOCH_34), &bytes(DAY_20731)).is_ok());
}
/// Known-mismatched: a well-formed pack for the previous epoch is "out of date" against the new one, in plain
/// words with both epochs named.
#[test]
fn known_mismatched_pack_is_out_of_date() {
let (h, s, _) = pack_texts(EPOCH_33, DAY_20731);
let err = verify_pack_texts(&h, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err();
assert!(matches!(err, PackFault::OutOfDate { .. }), "{err}");
assert_eq!(err.to_string(), "program pack out of date: the pack is for epoch bed7ab62cbece66c day 20731, the node is on epoch 009858237e118f69 day 20731");
}
/// A pack that contradicts itself is "disagree", never "out of date": seeds.txt of one epoch with program.h of
/// another (a half rewritten directory), or init words that are not the attempt's words (a worker on the old
/// rule would have produced this verdict for every retried program).
#[test]
fn inconsistent_pack_disagrees() {
let (h33, _, _) = pack_texts(EPOCH_33, DAY_20731);
let s34 = format!("epoch_seed_hex {EPOCH_34}\nday_seed_hex {DAY_20731}\n");
let err = verify_pack_texts(&h33, Some(&s34), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err();
assert!(matches!(err, PackFault::Disagree(_)), "{err}");
assert!(err.to_string().starts_with("program pack and its seeds disagree: seeds.txt names epoch 009858237e118f69"), "{err}");
let (h34, s, _) = pack_texts(EPOCH_34, DAY_20731);
let bare = attempt_words(&bytes(EPOCH_34), 0);
let edited = h34.lines().map(|l| if l.starts_with("#define IGNEUM_SEEDW_INIT") { format!("#define IGNEUM_SEEDW_INIT {{ {} }}", bare.iter().map(|w| format!("0x{w:08x}")).collect::<Vec<_>>().join(", ")) } else { l.to_string() }).collect::<Vec<_>>().join("\n");
let err = verify_pack_texts(&edited, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).unwrap_err();
assert!(err.to_string().contains("IGNEUM_SEEDW_INIT is not attempt"), "{err}");
// and a pack with no attempt line at all is read as attempt 0 (the packs before generator version 2)
let no_attempt = h34.lines().filter(|l| !l.starts_with("#define IGNEUM_PROGRAM_ATTEMPT")).collect::<Vec<_>>().join("\n");
assert!(verify_pack_texts(&no_attempt, Some(&s), &bytes(EPOCH_34), &bytes(DAY_20731)).is_err());
}
#[test]
fn day_label_reads_the_index() {
assert_eq!(day_label(DAY_20731), "20731");
assert_eq!(day_label("abcd"), "abcd");
}
/// Counter ASIC 2.0: a class v3 chain pack carries generator 3, the class line and the era seed; it is refused
/// when the chain wants class v2, when the era differs, and a v2 pack is refused when the chain wants v3; a
/// generator this software does not run is refused whatever is wanted.
#[test]
fn program_class_and_era_are_checked() {
let e = bytes(EPOCH_34);
let d = bytes(DAY_20731);
let era = [0x5au8; 32];
let v3 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V3, "class test");
let h3 = program_header(&v3.program, "test day", &v3.dataset);
assert!(h3.contains("#define IGNEUM_GENERATOR 3\n"));
assert!(h3.contains("#define IGNEUM_PROGRAM_CLASS \"v3\"\n"));
assert!(h3.contains(&format!("#define IGNEUM_ERA_SEED_HEX \"{}\"\n", hex(&era))));
let id = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V3), Some(&era)).unwrap();
assert_eq!((id.generator, id.class, id.era_hex.as_deref()), (3, ProgramClass::V3, Some(hex(&era).as_str())));
assert_eq!(id.attempt, v3.program.attempt);
assert!(verify_pack_texts(&h3, None, &e, &d).is_ok(), "no class wanted: either class passes");
let err = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V2), None).unwrap_err();
assert!(matches!(err, PackFault::WrongClass(_)), "{err}");
assert!(err.to_string().contains("program pack of the wrong class"), "{err}");
let err = verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V3), Some(&[1u8; 32])).unwrap_err();
assert!(err.to_string().contains("era seed"), "{err}");
// the v2 pack of the same seeds: generator 2, no class line, no era line, refused when v3 is wanted
let v2 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V2, "class test");
let h2 = program_header(&v2.program, "test day", &v2.dataset);
assert!(h2.contains("#define IGNEUM_GENERATOR 2\n"));
assert!(!h2.contains("IGNEUM_PROGRAM_CLASS") && !h2.contains("IGNEUM_ERA_SEED_HEX"));
let plain = Epoch::from_seed_bytes(&e, &d, "class test");
assert_eq!(program_header(&plain.program, "test day", &plain.dataset), h2, "class v2 from the chain is the v2 export byte for byte");
let id = verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V2), Some(&era)).unwrap();
assert_eq!((id.generator, id.class, id.era_hex), (2, ProgramClass::V2, None));
assert!(matches!(verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V3), None), Err(PackFault::WrongClass(_))));
// a generator nobody runs
let h9 = h2.replace("#define IGNEUM_GENERATOR 2\n", "#define IGNEUM_GENERATOR 9\n");
assert!(matches!(verify_pack_texts(&h9, None, &e, &d), Err(PackFault::WrongClass(_))));
// a class line that contradicts the generator
let bad = h3.replace("#define IGNEUM_PROGRAM_CLASS \"v3\"\n", "#define IGNEUM_PROGRAM_CLASS \"v2\"\n");
assert!(matches!(verify_pack_texts(&bad, None, &e, &d), Err(PackFault::Disagree(_))));
// Counter ASIC 3.0: a class v4 chain pack carries generator 4, the class line and the era seed; it is refused
// when v3 (or v2) is wanted, and a v3 pack is refused when v4 is wanted; the era rule applies to v4 as to v3
let v4 = Epoch::from_chain_seeds(&e, &d, Some(&era), ProgramClass::V4, "class test");
let h4 = program_header(&v4.program, "test day", &v4.dataset);
assert!(h4.contains("#define IGNEUM_GENERATOR 4\n"));
assert!(h4.contains("#define IGNEUM_PROGRAM_CLASS \"v4\"\n"));
assert!(h4.contains(&format!("#define IGNEUM_ERA_SEED_HEX \"{}\"\n", hex(&era))));
assert!(h4.contains("#define IGNEUM_SHADOW_INSTRS 256\n") && h4.contains("#define IGNEUM_SHADOW_REPS 27\n"), "{h4}");
let id = verify_pack_texts_chain(&h4, None, &e, &d, Some(ProgramClass::V4), Some(&era)).unwrap();
assert_eq!((id.generator, id.class, id.era_hex.as_deref()), (4, ProgramClass::V4, Some(hex(&era).as_str())));
assert_eq!(id.attempt, v3.program.attempt, "the v4 attempt is the v3 attempt of the same seed");
assert!(verify_pack_texts(&h4, None, &e, &d).is_ok(), "no class wanted: any class passes");
assert!(matches!(verify_pack_texts_chain(&h4, None, &e, &d, Some(ProgramClass::V3), None), Err(PackFault::WrongClass(_))));
assert!(matches!(verify_pack_texts_chain(&h4, None, &e, &d, Some(ProgramClass::V2), None), Err(PackFault::WrongClass(_))));
assert!(matches!(verify_pack_texts_chain(&h3, None, &e, &d, Some(ProgramClass::V4), None), Err(PackFault::WrongClass(_))));
assert!(matches!(verify_pack_texts_chain(&h2, None, &e, &d, Some(ProgramClass::V4), None), Err(PackFault::WrongClass(_))));
let err = verify_pack_texts_chain(&h4, None, &e, &d, Some(ProgramClass::V4), Some(&[1u8; 32])).unwrap_err();
assert!(err.to_string().contains("era seed"), "{err}");
let no_era = h4.replace(&format!("#define IGNEUM_ERA_SEED_HEX \"{}\"\n", hex(&era)), "");
let err = verify_pack_texts_chain(&no_era, None, &e, &d, Some(ProgramClass::V4), Some(&era)).unwrap_err();
assert!(err.to_string().contains("class v4 pack without IGNEUM_ERA_SEED_HEX"), "{err}");
// the v3 pack of the same seeds is unchanged by the v4 class existing
assert_eq!(program_header(&v3.program, "test day", &v3.dataset), h3);
// the 6 October trap: a v4 program stamped generator 3 (the CLI's old --era path) is refused for its shadow
// block, whatever class is wanted; and a generator 4 header without the block is refused too
let stamped3 = h4.replace("#define IGNEUM_GENERATOR 4\n", "#define IGNEUM_GENERATOR 3\n").replace("#define IGNEUM_PROGRAM_CLASS \"v4\"\n", "#define IGNEUM_PROGRAM_CLASS \"v3\"\n");
let err = verify_pack_texts(&stamped3, None, &e, &d).unwrap_err();
assert!(matches!(err, PackFault::Disagree(_)) && err.to_string().contains("shadow block"), "{err}");
assert!(verify_pack_texts_chain(&stamped3, None, &e, &d, Some(ProgramClass::V3), Some(&era)).is_err());
let no_shadow = h4.replace("#define IGNEUM_SHADOW_INSTRS 256\n", "");
let err = verify_pack_texts(&no_shadow, None, &e, &d).unwrap_err();
assert!(matches!(err, PackFault::Disagree(_)) && err.to_string().contains("without IGNEUM_SHADOW_INSTRS"), "{err}");
}
}

View file

@ -0,0 +1,114 @@
//! Seed derivation and the SplitMix64 stream, exactly as `proto-metal/main.swift` does them.
//!
//! Today a seed is a string ("igneum-genesis", "day/2026-10-03"). On the chain the epoch seed will be the
//! output of a class-group VDF over a certified checkpoint hash. [`seed_words_from_bytes`] is the function
//! boundary for that: whatever bytes the chain settles on go through the same FNV-1a construction, so the
//! generator and the day key never need to know where the bytes came from.
/// FNV-1a 64 over `bytes` with the standard basis. Used for the cache fingerprint in the packs.
pub fn fnv1a64(bytes: &[u8]) -> u64 {
fnv1a64_with_basis(0xcbf29ce484222325, bytes)
}
#[inline]
fn fnv1a64_with_basis(basis: u64, bytes: &[u8]) -> u64 {
let mut h = basis;
for &b in bytes {
h ^= b as u64;
h = h.wrapping_mul(0x100000001b3);
}
h
}
/// FNV-1a 64 over 32-bit words in little-endian byte order (the cache is hashed as raw memory).
pub fn fnv1a64_words(words: &[u32]) -> u64 {
let mut h: u64 = 0xcbf29ce484222325;
for &w in words {
for b in w.to_le_bytes() {
h ^= b as u64;
h = h.wrapping_mul(0x100000001b3);
}
}
h
}
/// The 32-byte seed (8 x u32) from arbitrary bytes: FNV-1a 64 with four salts, each finalised with the
/// murmur-style mix `h ^= h >> 33; h *= 0xff51afd7ed558ccd; h ^= h >> 33`; low word then high word.
pub fn seed_words_from_bytes(bytes: &[u8]) -> [u32; 8] {
let mut words = [0u32; 8];
for salt in 0..4u64 {
let basis = 0xcbf29ce484222325u64 ^ salt.wrapping_mul(0x9E3779B97F4A7C15);
let mut h = fnv1a64_with_basis(basis, bytes);
h ^= h >> 33;
h = h.wrapping_mul(0xff51afd7ed558ccd);
h ^= h >> 33;
words[2 * salt as usize] = h as u32;
words[2 * salt as usize + 1] = (h >> 32) as u32;
}
words
}
/// The 32-byte seed from a string (its UTF-8 bytes). `seedWords` in the Swift.
pub fn seed_words(s: &str) -> [u32; 8] {
seed_words_from_bytes(s.as_bytes())
}
/// The day key: the 8 words of `seed_words("day/" + day)`. `K` in MEMHARD.md; `d0, d1` are `K[0], K[1]`.
pub fn day_key(day: &str) -> [u32; 8] {
seed_words(&format!("day/{day}"))
}
/// SplitMix64, the one deterministic stream every draw in the prototype comes from.
#[derive(Clone, Copy, Debug)]
pub struct SplitMix64 {
pub s: u64,
}
impl SplitMix64 {
pub fn new(s: u64) -> Self {
Self { s }
}
#[inline]
pub fn next(&mut self) -> u64 {
self.s = self.s.wrapping_add(0x9E3779B97F4A7C15);
let mut z = self.s;
z = (z ^ (z >> 30)).wrapping_mul(0xBF58476D1CE4E5B9);
z = (z ^ (z >> 27)).wrapping_mul(0x94D049BB133111EB);
z ^ (z >> 31)
}
/// `next() % n` as the Swift `below` does it (modulo, not rejection sampling).
#[inline]
pub fn below(&mut self, n: u64) -> u64 {
self.next() % n
}
}
/// The generator's stream for a seed: `(w0 | w1 << 32) ^ ((w2 | w3 << 32) * 0x9E3779B97F4A7C15)`.
pub fn program_rng(seed: &[u32; 8]) -> SplitMix64 {
let lo = seed[0] as u64 | ((seed[1] as u64) << 32);
let hi = seed[2] as u64 | ((seed[3] as u64) << 32);
SplitMix64::new(lo ^ hi.wrapping_mul(0x9E3779B97F4A7C15))
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn genesis_seed_words() {
// From proto-cuda/packs/igneum-genesis/program.json.
assert_eq!(
seed_words("igneum-genesis"),
[0x67a9a7be, 0x1a155b25, 0xfddfb732, 0x4b5af2e8, 0xc55caf33, 0xa27c13b7, 0x06628a48, 0x03852469]
);
}
#[test]
fn day_key_2026_10_03() {
// MEMHARD.md section 1.1.
assert_eq!(
day_key("2026-10-03"),
[0x3067619f, 0x3c269176, 0x84a03b03, 0xf8c63294, 0xff977c5b, 0xe60def3e, 0x63630141, 0xb8fbcb58]
);
}
}

View file

@ -0,0 +1,287 @@
//! Class v5, proof of stored state and of following (`docs/design/class-v5-stored-state.md`, 7 October 2026): the
//! leaves the item derivation XORs in (section 2 of the page), built from the canonical state stream of the
//! window's reference block.
//!
//! `D[i] = Blake2b-512("igneum-sd1/" || root || i_le32 || record_i)` for the `n` records of the stream, and item
//! `t` takes `leaf(t) = D[t mod n]`: every item is keyed by the state, so a hasher without it is wrong on every
//! item (the known-failed case, the first test). When the stream has more records than the dataset has items, the
//! records are ordered by `Blake2b-256("igneum-sd1-sample/" || root || record)` and the first `items` are taken, a
//! sample nobody can choose without the whole state and the root.
//!
//! The stream file (`StateStream`): the plain format every side reads without a serialisation library, `IGSD1\0`,
//! the chain block number (le64) and hash (32), the state root (32), the record count (le32), then each record as
//! its length (le32) and bytes. The node's executor writes it (`igneum/exec/src/day_stream.rs`), the miner fetches
//! it, the CLI's `--state` reads it, and a pack carries the leaves it yields as `leaves.bin`.
use crate::blake2b::{blake2b_256, blake2b_512};
use crate::seed::fnv1a64_words;
pub const LEAF_TAG: &[u8] = b"igneum-sd1/";
pub const SAMPLE_TAG: &[u8] = b"igneum-sd1-sample/";
pub const STREAM_MAGIC: &[u8; 6] = b"IGSD1\0";
/// The canonical state stream at one chain block: what the executor serialises and what the leaves derive from.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct StateStream {
pub number: u64,
pub block: [u8; 32],
pub root: [u8; 32],
pub records: Vec<Vec<u8>>,
}
impl StateStream {
pub fn encode(&self) -> Vec<u8> {
let mut b = Vec::with_capacity(6 + 8 + 32 + 32 + 4 + self.records.iter().map(|r| 4 + r.len()).sum::<usize>());
b.extend_from_slice(STREAM_MAGIC);
b.extend_from_slice(&self.number.to_le_bytes());
b.extend_from_slice(&self.block);
b.extend_from_slice(&self.root);
b.extend_from_slice(&(self.records.len() as u32).to_le_bytes());
for r in &self.records {
b.extend_from_slice(&(r.len() as u32).to_le_bytes());
b.extend_from_slice(r);
}
b
}
pub fn decode(bytes: &[u8]) -> Result<StateStream, String> {
if bytes.len() < 6 + 8 + 32 + 32 + 4 || &bytes[..6] != STREAM_MAGIC {
return Err("not a state stream file (magic IGSD1)".into());
}
let mut at = 6;
let number = u64::from_le_bytes(bytes[at..at + 8].try_into().unwrap());
at += 8;
let block: [u8; 32] = bytes[at..at + 32].try_into().unwrap();
at += 32;
let root: [u8; 32] = bytes[at..at + 32].try_into().unwrap();
at += 32;
let n = u32::from_le_bytes(bytes[at..at + 4].try_into().unwrap()) as usize;
at += 4;
let mut records = Vec::with_capacity(n.min(1 << 20));
for i in 0..n {
if at + 4 > bytes.len() {
return Err(format!("state stream truncated at record {i} of {n}"));
}
let len = u32::from_le_bytes(bytes[at..at + 4].try_into().unwrap()) as usize;
at += 4;
if at + len > bytes.len() {
return Err(format!("state stream truncated inside record {i} of {n}"));
}
records.push(bytes[at..at + len].to_vec());
at += len;
}
if at != bytes.len() {
return Err(format!("state stream has {} trailing bytes", bytes.len() - at));
}
Ok(StateStream { number, block, root, records })
}
pub fn read_file(path: &std::path::Path) -> Result<StateStream, String> {
let bytes = std::fs::read(path).map_err(|e| format!("read {}: {e}", path.display()))?;
Self::decode(&bytes)
}
}
/// `D[i]`: the 64-byte digest of record `i` under `root`, as 16 little-endian words.
pub fn leaf_digest(root: &[u8; 32], i: u32, record: &[u8]) -> [u32; 16] {
let d = blake2b_512(&[LEAF_TAG, root, &i.to_le_bytes(), record]);
let mut w = [0u32; 16];
for (k, x) in w.iter_mut().enumerate() {
*x = u32::from_le_bytes(d[k * 4..k * 4 + 4].try_into().unwrap());
}
w
}
/// The sample order key of a record under `root`.
pub fn sample_key(root: &[u8; 32], record: &[u8]) -> [u8; 32] {
blake2b_256(&[SAMPLE_TAG, root, record])
}
/// The leaves of one window (or day) of class v5: `n` digests of 64 bytes, `leaf(t) = D[t mod n]`.
#[derive(Clone, Debug, PartialEq, Eq)]
pub struct StateLeaves {
pub root: [u8; 32],
pub block: [u8; 32],
pub number: u64,
/// Records in the stream before any sample.
pub records_total: u64,
/// Whether the stream had more records than the dataset has items (the sample rule applied).
pub sampled: bool,
leaves: Vec<[u32; 16]>,
}
impl StateLeaves {
/// The items a dataset of `2^log2_words` words has: `2^(log2_words - 4)`.
pub fn items_of(log2_words: u32) -> u64 {
1u64 << log2_words.saturating_sub(4)
}
/// The leaves of `records` (canonical order) under `root` for a dataset of `2^log2_words` words. An empty stream
/// yields one leaf, the digest of the empty record, so `n` is never 0.
pub fn build(root: [u8; 32], block: [u8; 32], number: u64, records: &[Vec<u8>], log2_words: u32) -> StateLeaves {
let items = Self::items_of(log2_words);
let records_total = records.len() as u64;
let empty: Vec<Vec<u8>> = vec![Vec::new()];
let records = if records.is_empty() { &empty[..] } else { records };
let sampled = records.len() as u64 > items;
let chosen: Vec<&Vec<u8>> = if sampled {
let mut keyed: Vec<([u8; 32], &Vec<u8>)> = records.iter().map(|r| (sample_key(&root, r), r)).collect();
keyed.sort_unstable_by(|a, b| a.0.cmp(&b.0).then_with(|| a.1.cmp(b.1)));
keyed.into_iter().take(items as usize).map(|(_, r)| r).collect()
} else {
records.iter().collect()
};
let leaves = chosen.iter().enumerate().map(|(i, r)| leaf_digest(&root, i as u32, r)).collect();
StateLeaves { root, block, number, records_total, sampled, leaves }
}
pub fn from_stream(s: &StateStream, log2_words: u32) -> StateLeaves {
Self::build(s.root, s.block, s.number, &s.records, log2_words)
}
/// Leaves from the raw words of a `leaves.bin` (16 words per leaf), for a worker or a test that holds no stream.
pub fn from_words(root: [u8; 32], block: [u8; 32], number: u64, words: &[u32]) -> StateLeaves {
assert!(!words.is_empty() && words.len() % 16 == 0, "leaves are 16 words each");
let leaves = words.chunks_exact(16).map(|c| c.try_into().unwrap()).collect::<Vec<[u32; 16]>>();
StateLeaves { root, block, number, records_total: leaves.len() as u64, sampled: false, leaves }
}
#[inline(always)]
pub fn n(&self) -> u32 {
self.leaves.len() as u32
}
/// `leaf(t) = D[t mod n]`.
#[inline(always)]
pub fn leaf(&self, t: u32) -> &[u32; 16] {
&self.leaves[(t % self.n()) as usize]
}
pub fn leaves(&self) -> &[[u32; 16]] {
&self.leaves
}
/// The flat words of `leaves.bin`.
pub fn words(&self) -> Vec<u32> {
self.leaves.iter().flat_map(|l| l.iter().copied()).collect()
}
/// The bytes of `leaves.bin` (little-endian words).
pub fn bytes(&self) -> Vec<u8> {
self.words().iter().flat_map(|w| w.to_le_bytes()).collect()
}
/// FNV-1a 64 over the leaves as little-endian bytes (the pack's `IGNEUM_STATE_LEAVES_FNV64`).
pub fn fnv1a64(&self) -> u64 {
fnv1a64_words(&self.words())
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::generator::{generate_class, V5_CLASS};
use crate::memhard::{derive_item_leaves, Cache, MixParams, Shape};
use crate::seed::day_key;
use crate::verify::{hash_warp, DatasetMode, DatasetSource};
use std::sync::Arc;
fn records(n: usize, salt: u8) -> Vec<Vec<u8>> {
(0..n).map(|i| vec![salt, i as u8, (i >> 8) as u8, 7]).collect()
}
fn leaves(root: u8, n: usize, log2_words: u32) -> Arc<StateLeaves> {
Arc::new(StateLeaves::build([root; 32], [0x22; 32], 5, &records(n, root), log2_words))
}
/// The known-failed case, first: a hasher without the state (no leaves, the leaves of another root, the leaves
/// of a stream one record short, the previous window's leaves) is wrong on every item and every lane.
#[test]
fn a_stateless_hasher_is_wrong_on_every_item() {
let key = day_key("2026-10-03");
let shape = Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: true };
let cache = Arc::new(Cache::fill_log2(key, 16));
let mp = MixParams::with_shape(key, shape);
let good = leaves(0x11, 93, 20);
let other_root = leaves(0x12, 93, 20);
let one_short = Arc::new(StateLeaves::build([0x13; 32], [0x22; 32], 5, &records(92, 0x11), 20)); // a record short means another root
let previous_window = leaves(0x10, 93, 20);
for (name, bad) in [("another root", other_root.clone()), ("one record short", one_short.clone()), ("the previous window", previous_window.clone())] {
let equal = (0..64u32).filter(|&t| derive_item_leaves(t * 7919, &mp, &cache, Some(&good)) == derive_item_leaves(t * 7919, &mp, &cache, Some(&bad))).count();
assert_eq!(equal, 0, "{name}: {equal} of 64 items equal");
}
let stateless = Shape { state: false, ..shape };
let mp_stateless = MixParams::with_shape(key, stateless);
let equal = (0..64u32).filter(|&t| derive_item_leaves(t * 7919, &mp, &cache, Some(&good)) == derive_item_leaves(t * 7919, &mp_stateless, &cache, None)).count();
assert_eq!(equal, 0, "no leaves at all: {equal} of 64 items equal");
// the warp: a class v5 program over a small dataset, the same program and cache, other leaves
let program = generate_class("igneum-genesis", V5_CLASS);
let ds = DatasetSource::new_shape("2026-10-03", DatasetMode::MemoryHard, 20, shape).with_leaves(good.clone());
let ds_other = DatasetSource::new_shape("2026-10-03", DatasetMode::MemoryHard, 20, shape).with_leaves(previous_window.clone());
let a = hash_warp(&program, 0, &ds);
let b = hash_warp(&program, 0, &ds_other);
assert_eq!(a.iter().zip(b.iter()).filter(|(x, y)| x == y).count(), 0, "0 of 32 lanes agree");
assert_eq!(hash_warp(&program, 0, &ds), a, "the same leaves hash the same");
}
/// Every item takes a leaf: `leaf(t) = D[t mod n]`, so items `t` and `t + n` share a leaf and still differ.
#[test]
fn every_item_is_keyed_and_the_leaf_wraps() {
let l = leaves(0x11, 93, 28);
assert_eq!(l.n(), 93);
assert!(!l.sampled);
assert_eq!(l.records_total, 93);
for t in [0u32, 1, 92, 93, 94, 1_000_000, u32::MAX] {
assert_eq!(l.leaf(t), l.leaf(t % 93));
assert_eq!(*l.leaf(t), leaf_digest(&[0x11; 32], t % 93, &records(93, 0x11)[(t % 93) as usize]));
}
let key = day_key("2026-10-03");
let shape = Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: true };
let cache = Cache::fill_log2(key, 16);
let mp = MixParams::with_shape(key, shape);
assert_ne!(derive_item_leaves(5, &mp, &cache, Some(&l)), derive_item_leaves(5 + 93, &mp, &cache, Some(&l)));
// an empty stream yields one leaf (the digest of the empty record), never a division by zero
let empty = StateLeaves::build([0x11; 32], [0; 32], 0, &[], 28);
assert_eq!(empty.n(), 1);
assert_eq!(empty.records_total, 0);
assert_eq!(*empty.leaf(12_345), leaf_digest(&[0x11; 32], 0, &[]));
}
/// Above the dataset size the records are sampled in the keyed order: a different root picks a different set,
/// and the set cannot be the first `items` records of the stream.
#[test]
fn the_sample_above_the_dataset_size_is_keyed_by_the_root() {
let recs = records(40, 0x33);
let a = StateLeaves::build([0x11; 32], [0; 32], 0, &recs, 8);
let b = StateLeaves::build([0x12; 32], [0; 32], 0, &recs, 8);
assert_eq!(StateLeaves::items_of(8), 16);
assert_eq!((a.n(), a.sampled, a.records_total), (16, true, 40));
assert_ne!(a.leaves(), b.leaves(), "another root, another sample");
// the positional first 16 are not the sample (with overwhelming probability for 40 choose 16)
let positional = StateLeaves::build([0x11; 32], [0; 32], 0, &recs[..16], 8);
assert_ne!(a.leaves(), positional.leaves());
// the same inputs sample the same
assert_eq!(StateLeaves::build([0x11; 32], [0; 32], 0, &recs, 8), a);
// at the dataset size exactly, no sample
let c = StateLeaves::build([0x11; 32], [0; 32], 0, &recs[..16], 8);
assert!(!c.sampled && c.n() == 16);
}
#[test]
fn stream_file_round_trip_and_refusals() {
let s = StateStream { number: 159_357, block: [0xaf; 32], root: [0x1c; 32], records: records(93, 1) };
let bytes = s.encode();
assert_eq!(&bytes[..6], STREAM_MAGIC);
assert_eq!(StateStream::decode(&bytes).unwrap(), s);
assert!(StateStream::decode(&bytes[..bytes.len() - 1]).is_err(), "truncated");
let mut trailing = bytes.clone();
trailing.push(0);
assert!(StateStream::decode(&trailing).is_err(), "trailing bytes");
assert!(StateStream::decode(b"IGSD0\0").is_err(), "wrong magic");
let l = StateLeaves::from_stream(&s, 28);
let back = StateLeaves::from_words(s.root, s.block, s.number, &l.words());
assert_eq!(back.leaves(), l.leaves());
assert_eq!(l.bytes().len(), 93 * 64);
assert_eq!(l.fnv1a64(), back.fnv1a64());
}
}

View file

@ -0,0 +1,935 @@
//! The CPU reference interpreter for one 32-lane warp (`cpuWarpTraced` in the Swift) and the API the node
//! calls. Dataset words come from the memory-hard cache (default) or from the closed form (old packs).
use crate::generator::{generate, generate_class, EraParams, Instr, LoadClass, Op, Program, ProgramClass, ITERATIONS, LANES};
use crate::memhard::{hot_index, HotTable, Layout, MemhardCpu, Shape};
use crate::seed::day_key;
/// The load address of an era program (`docs/plans/era-layout.md` section 1.3): `y = rotl(x * M, R)`, then the
/// window of the load site, `k = min(win, D - 26)` (0 when `D <= 26`), `idx = ((y & (MASK >> k)) | ((off &
/// (2^k - 1)) << (D - k))) & MASK`. For every other class `idx = x & MASK`, the lottery hash's address. `mask` is
/// `2^D - 1`. The acceptance mirror calls this at the rule's constant `D = 28`.
#[inline(always)]
pub fn load_index(era: Option<&EraParams>, ins: &Instr, x: u32, mask: u32, log2: u32) -> u32 {
match era {
None => x & mask,
Some(e) => {
let (wm, off) = window(ins, mask, log2);
let y = x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot);
((y & wm) | off) & mask
}
}
}
/// The window of a load site at a dataset of `2^log2` words: `(window mask, offset)` such that
/// `idx = (y & window mask) | offset` lies in the site's aligned window of `2^(log2 - k)` words.
#[inline(always)]
pub fn window(ins: &Instr, mask: u32, log2: u32) -> (u32, u32) {
let k = (ins.win as u32).min(log2.saturating_sub(26));
let wm = mask >> k;
let off = ((ins.off as u32) & ((1u32 << k) - 1)) << (log2 - k);
(wm, off)
}
/// Read-width experiment (5 October 2026): a `load` of `W` words folds every word into `dst`:
/// `x = dst XOR w[0]; for j in 1..W: x = (rotl(x, FOLD_ROT) * FOLD_MUL) XOR w[j]; dst = x`. For `W = 1` this is the
/// lottery hash's `dst XOR dataset[...]`. The fold is state-dependent (the rotate-multiply sits between the words),
/// so no function of the line alone replaces it: two different lines give two different maps of `dst`, and a
/// dataset of folded lines cannot be stored in place of the dataset (see `docs/plans/read-width.md`).
pub const FOLD_ROT: u32 = 11;
pub const FOLD_MUL: u32 = 0x9E3779B1;
/// The fold of `words` into `dst` (at least one word).
#[inline(always)]
pub fn fold_words(dst: u32, words: &[u32]) -> u32 {
let mut x = dst ^ words[0];
for &w in &words[1..] {
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w;
}
x
}
/// Variant 5 (scratch): the fill value of word `j` (0..2) of slot `slot` of lane `lane` of the unit at base nonce
/// `base`, under program seed words `seed`. The scratch of a unit starts as these values; a slot written during
/// the unit's hash holds what was written. Mirrored as `scr_fill` in every emitted kernel.
#[inline(always)]
pub fn scratch_fill(seed: &[u32; 8], base: u32, lane: u32, slot: u32, j: u32) -> u32 {
splitmix32(
(base.wrapping_add(lane) ^ seed[j as usize])
.wrapping_add(slot.wrapping_mul(0x9E3779B1))
.wrapping_add((j + 1).wrapping_mul(0x85EBCA77)),
)
}
/// Variant 5: the 16-byte slot after a read-modify-write that read `w` and folded to `x`: `(x ^ w1, rotl(x, 7) ^ w2,
/// x + w0)` behind the slot's tag.
#[inline(always)]
pub fn scratch_rewrite(x: u32, w: &[u32; 3]) -> [u32; 3] {
[x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])]
}
/// The CPU model of one unit's scratch (variant 5): per lane, the written slots and their words. Unwritten slots
/// read as [`scratch_fill`]. A unit touches at most `scratch ops x 32` slots; a GPU keeps the real scratch per
/// resident warp with a per-unit tag per slot.
pub struct ScratchModel {
slots: usize,
written: Vec<bool>,
data: Vec<[u32; 3]>,
pub reads: usize,
pub writes: usize,
/// Soundness tests (`tests/scratch.rs`, `docs/analysis/scratch-soundness.md`): when `Some`, every
/// read-modify-write is appended as it happened. `None` on every verification path.
pub trace: Option<Vec<ScratchEvent>>,
}
/// One scratch read-modify-write as the interpreter saw it (variant 5 soundness tests).
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub struct ScratchEvent {
pub lane: u8,
pub slot: u32,
/// The slot had been written earlier in this unit (a re-hit): the words read were a rewrite, not the fill.
pub hit: bool,
pub read: [u32; 3],
/// The fold result, the new value of `dst`.
pub x: u32,
pub written: [u32; 3],
}
impl ScratchModel {
pub fn new(slots_per_lane: usize) -> Self {
Self {
slots: slots_per_lane,
written: vec![false; LANES * slots_per_lane],
data: vec![[0; 3]; LANES * slots_per_lane],
reads: 0,
writes: 0,
trace: None,
}
}
/// Read slot `slot` of `lane`, then rewrite it from the fold result `x`. Returns the three words read.
#[inline]
pub fn rmw(&mut self, seed: &[u32; 8], base: u32, lane: usize, slot: u32, dst: u32) -> u32 {
let i = lane * self.slots + slot as usize;
let w = if self.written[i] {
self.data[i]
} else {
[
scratch_fill(seed, base, lane as u32, slot, 0),
scratch_fill(seed, base, lane as u32, slot, 1),
scratch_fill(seed, base, lane as u32, slot, 2),
]
};
let x = fold_words(dst, &w);
let out = scratch_rewrite(x, &w);
if let Some(t) = self.trace.as_mut() {
t.push(ScratchEvent { lane: lane as u8, slot, hit: self.written[i], read: w, x, written: out });
}
self.data[i] = out;
self.written[i] = true;
self.reads += 1;
self.writes += 1;
x
}
}
/// Dataset element, closed form of (day words, index). The original prototype's six-operation element.
#[inline(always)]
pub fn dataset_elem(i: u32, d0: u32, d1: u32) -> u32 {
let mut x = i ^ d0;
x = x.wrapping_mul(0x9E3779B1);
x ^= x >> 15;
x = x.wrapping_add(d1);
x = x.wrapping_mul(0x85EBCA77);
x ^= x >> 13;
x = x.wrapping_mul(0xC2B2AE3D);
x ^= x >> 16;
x
}
/// splitmix32, used for the register init.
#[inline(always)]
pub fn splitmix32(v: u32) -> u32 {
let mut x = v;
x ^= x >> 16;
x = x.wrapping_mul(0x7feb352d);
x ^= x >> 15;
x = x.wrapping_mul(0x846ca68b);
x ^= x >> 16;
x
}
/// Which construction fills the dataset words.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum DatasetMode {
/// The six-operation closed form (packs igneum-genesis and igneum-hourly). Not memory-hard.
ClosedForm,
/// The 256 MiB cache and 8 dependent reads per item (pack igneum-genesis-mh, MEMHARD.md). The default.
MemoryHard,
}
impl DatasetMode {
pub fn name(self) -> &'static str {
match self {
DatasetMode::ClosedForm => "closed-form",
DatasetMode::MemoryHard => "memory-hard",
}
}
}
/// Where the interpreter reads dataset words from.
pub enum Dataset {
ClosedForm { d0: u32, d1: u32 },
MemoryHard(MemhardCpu),
}
/// A dataset of `2^log2` words plus the construction that fills it.
pub struct DatasetSource {
pub log2_words: u32,
pub mask: u32,
/// The day key `K`; `d0, d1 = K[0], K[1]`.
pub key: [u32; 8],
/// The bytes `K` was derived from (`"day/<day>"` for a string day, `bind::day_bytes` on the chain), recorded
/// in packs so any implementation can rebuild the key. Empty when the key was given directly.
pub key_bytes: Vec<u8>,
pub dataset: Dataset,
/// The hot table of the epoch (hot-table experiment, `docs/plans/hot-table.md`): `Some` when the program's
/// class has one; filled by [`Epoch::new_class`] and [`Epoch::from_seed_bytes_class`] from the program's seed
/// bytes. A hot load reads `hot[hot_index(src, words)]`.
pub hot: Option<HotTable>,
}
impl DatasetSource {
/// Build the source for a day. Memory-hard mode fills the 256 MiB cache on the calling thread.
pub fn new(day: &str, mode: DatasetMode, log2_words: u32) -> Self {
Self::new_shape(day, mode, log2_words, Shape::V2)
}
/// [`DatasetSource::new`] with the construction's shape (mixer multiplier, cache size; Counter ASIC 2.0).
pub fn new_shape(day: &str, mode: DatasetMode, log2_words: u32, shape: Shape) -> Self {
let mut ds = Self::from_key_shape(day_key(day), mode, log2_words, shape);
ds.key_bytes = format!("day/{day}").into_bytes();
ds
}
pub fn from_key(key: [u32; 8], mode: DatasetMode, log2_words: u32) -> Self {
Self::from_key_shape(key, mode, log2_words, Shape::V2)
}
/// [`DatasetSource::from_key`] with the construction's shape. Memory-hard mode fills a cache of
/// `2^shape.cache_log2_words` words on the calling thread.
pub fn from_key_shape(key: [u32; 8], mode: DatasetMode, log2_words: u32, shape: Shape) -> Self {
assert!((4..=32).contains(&log2_words), "dataset log2 must be in 4..=32");
let mask = if log2_words == 32 { u32::MAX } else { (1u32 << log2_words) - 1 };
let dataset = match mode {
DatasetMode::ClosedForm => Dataset::ClosedForm { d0: key[0], d1: key[1] },
DatasetMode::MemoryHard => Dataset::MemoryHard(MemhardCpu::with_shape(key, shape)),
};
Self { log2_words, mask, key, key_bytes: Vec::new(), dataset, hot: None }
}
/// This source with the window's state leaves (class v5, `docs/design/class-v5-stored-state.md`): memory-hard mode
/// under a shape with `state` only.
pub fn with_leaves(mut self, leaves: std::sync::Arc<crate::state::StateLeaves>) -> Self {
match &mut self.dataset {
Dataset::MemoryHard(m) => {
assert!(m.params.shape.state, "state leaves on a dataset whose shape has no state");
m.leaves = Some(leaves);
}
Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"),
}
self
}
/// A source of the same day with other leaves, the 256 MiB cache shared (the class v5 window refresh).
pub fn refreshed(&self, leaves: std::sync::Arc<crate::state::StateLeaves>) -> Self {
let dataset = match &self.dataset {
Dataset::MemoryHard(m) => Dataset::MemoryHard(m.refreshed(leaves)),
Dataset::ClosedForm { .. } => panic!("state leaves on a closed-form dataset"),
};
Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None }
}
/// A copy of this source sharing its cache (and leaves), for a caller that needs an owned source from a shared one.
pub fn refreshed_or_clone(&self) -> Self {
let dataset = match &self.dataset {
Dataset::MemoryHard(m) => Dataset::MemoryHard(crate::memhard::MemhardCpu { params: m.params.clone(), cache: m.cache.clone(), leaves: m.leaves.clone() }),
Dataset::ClosedForm { d0, d1 } => Dataset::ClosedForm { d0: *d0, d1: *d1 },
};
Self { log2_words: self.log2_words, mask: self.mask, key: self.key, key_bytes: self.key_bytes.clone(), dataset, hot: None }
}
/// The window's state leaves, when the source carries them.
pub fn leaves(&self) -> Option<&std::sync::Arc<crate::state::StateLeaves>> {
self.memhard().and_then(|m| m.leaves.as_ref())
}
/// This source with the hot table of the epoch whose program seed bytes are `seed_bytes` (`mb` MiB).
pub fn with_hot(mut self, seed_bytes: &[u8], mb: u32) -> Self {
self.hot = Some(HotTable::for_seed_bytes(seed_bytes, mb));
self
}
/// The hot table of a program's class, filled from its seed bytes (none for a class without one).
pub fn attach_hot_for(&mut self, program: &Program) {
self.hot = program.class.hot.map(|h| HotTable::for_seed_bytes(&program.seed_bytes, h.mb as u32));
}
/// The shape of the memory-hard construction ([`Shape::V2`] for the closed form, which has none).
pub fn shape(&self) -> Shape {
self.memhard().map(|m| m.shape()).unwrap_or(Shape::V2)
}
pub fn mode(&self) -> DatasetMode {
match self.dataset {
Dataset::ClosedForm { .. } => DatasetMode::ClosedForm,
Dataset::MemoryHard(_) => DatasetMode::MemoryHard,
}
}
pub fn memhard(&self) -> Option<&MemhardCpu> {
match &self.dataset {
Dataset::MemoryHard(m) => Some(m),
_ => None,
}
}
/// `dataset[w & mask]` under the linear layout (the lottery hash).
pub fn word(&self, w: u32) -> u32 {
self.word_at(Layout::LINEAR, w)
}
/// `dataset[w & mask]` under a program's layout (era layout). The closed form has no items and ignores it.
pub fn word_at(&self, layout: Layout, w: u32) -> u32 {
let w = w & self.mask;
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => dataset_elem(w, *d0, *d1),
Dataset::MemoryHard(m) => m.word_at(layout, w),
}
}
/// `out[k] = dataset[idx[k]]`; indices are already masked. Returns items derived (0 for the closed form).
#[inline]
fn fetch(&self, idx: &[u32; LANES], out: &mut [u32; LANES], layout: Layout) -> usize {
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => {
for k in 0..LANES {
out[k] = dataset_elem(idx[k], *d0, *d1);
}
0
}
Dataset::MemoryHard(m) => m.fetch(idx, out, layout),
}
}
/// `out[k][j] = dataset[base[k] + j]` for `j < width`; bases are masked and aligned to `width` words
/// (`width` 4 or 16, so a lane's words lie in one item). Returns items derived (0 for the closed form).
#[inline]
fn fetch_wide(&self, base: &[u32; LANES], width: usize, out: &mut [[u32; 16]; LANES], layout: Layout) -> usize {
match &self.dataset {
Dataset::ClosedForm { d0, d1 } => {
for k in 0..LANES {
for j in 0..width {
out[k][j] = dataset_elem(base[k] + j as u32, *d0, *d1);
}
}
0
}
Dataset::MemoryHard(m) => m.fetch_wide(base, width, out, layout),
}
}
}
/// The result of interpreting one warp.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub struct WarpResult {
pub hashes: [u64; LANES],
/// Distinct dataset items derived from the cache (0 in closed-form mode).
pub items_derived: usize,
}
#[inline(always)]
fn mulhi32(a: u32, b: u32) -> u32 {
((a as u64 * b as u64) >> 32) as u32
}
/// Interpret `program` for the 32 nonces `base_nonce .. base_nonce + 31` (wrapping). Registers are kept
/// register-major (`r[reg][lane]`) so the per-lane loops vectorise; the semantics are those of the Metal
/// and CUDA kernels instruction for instruction.
pub fn interpret_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) -> WarpResult {
interpret_warp_init(program, &program.seed, base_nonce, ds)
}
/// [`interpret_warp`] with explicit init words `I` (section 1.6 of the spec). The packs use `I = program.seed`;
/// a block uses `I = bind::block_init_words(H, nonce)`.
pub fn interpret_warp_init(program: &Program, seed: &[u32; 8], base_nonce: u32, ds: &DatasetSource) -> WarpResult {
interpret_warp_scratch(program, seed, base_nonce, ds, false).0
}
/// [`interpret_warp_init`] that also returns every scratch read-modify-write of the unit in execution order
/// (lane-minor within an instruction, as the interpreter runs them) when `trace` is set; empty otherwise and for
/// a class without a scratch. For the soundness tests of variant 5 only.
pub fn interpret_warp_scratch(
program: &Program,
seed: &[u32; 8],
base_nonce: u32,
ds: &DatasetSource,
trace: bool,
) -> (WarpResult, Vec<ScratchEvent>) {
let mask = ds.mask;
let log2 = ds.log2_words;
let era = program.class.era;
let layout = program.class.layout();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
let nonce = base_nonce.wrapping_add(lane as u32);
for i in 0..8 {
let mut x = nonce ^ seed[i];
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
x = splitmix32(x);
r[i][lane] = x ^ seed[(i + 1) & 7];
}
}
let mut items_derived = 0usize;
let mut idx = [0u32; LANES];
let mut val = [0u32; LANES];
let mut scratch = if program.has_scratch() { Some(ScratchModel::new(program.class.scratch_slots_per_lane())) } else { None };
if trace {
if let Some(m) = scratch.as_mut() {
m.trace = Some(Vec::new());
}
}
let slot_mask = program.class.scratch_slot_mask();
if program.has_hot() {
let h = ds.hot.as_ref().expect("a hot-table program needs the epoch's hot table on the dataset source");
assert_eq!(h.n_words(), program.hot_words(), "the hot table's size is the class's");
}
for _ in 0..ITERATIONS {
let sel = r[0];
for ins in &program.instrs {
step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived);
if ins.op == Op::Scratch {
let m = scratch.as_mut().expect("a scratch op needs a scratch class");
let (d, a) = (ins.dst as usize, ins.src as usize);
for lane in 0..LANES {
let slot = r[a][lane] & slot_mask;
r[d][lane] = m.rmw(&program.seed, base_nonce, lane, slot, r[d][lane]);
}
}
}
// Latency-shadow block (Counter ASIC 3.0 item 8): the block runs `reps` times after instruction 63 with the
// iteration's `sel`; it is empty on every class without a shadow, so version 2 and class v3 run nothing here.
for _ in 0..program.shadow_reps() {
for ins in &program.shadow {
step(ins, &mut r, &sel, mask, log2, era.as_ref(), layout, ds, &mut idx, &mut val, &mut items_derived);
}
}
}
let mut hashes = [0u64; LANES];
for lane in 0..LANES {
let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21);
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
hashes[lane] = ((hi as u64) << 32) | lo as u64;
}
let events = scratch.and_then(|m| m.trace).unwrap_or_default();
(WarpResult { hashes, items_derived }, events)
}
#[inline(always)]
#[allow(clippy::too_many_arguments)]
fn step(
ins: &Instr,
r: &mut [[u32; LANES]; 8],
sel: &[u32; LANES],
mask: u32,
log2: u32,
era: Option<&EraParams>,
layout: Layout,
ds: &DatasetSource,
idx: &mut [u32; LANES],
val: &mut [u32; LANES],
items_derived: &mut usize,
) {
let d = ins.dst as usize;
let a = ins.src as usize;
match ins.op {
Op::Add => {
let (imm, imm2, bit) = (ins.imm, ins.imm2, ins.bit as u32);
let src = r[a];
for lane in 0..LANES {
let s = (sel[lane] >> bit) & 1;
let c = if s != 0 { imm2 } else { imm };
r[d][lane] = r[d][lane].wrapping_add(src[lane]).wrapping_add(c);
}
}
Op::Sub => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_sub(src[lane]);
}
}
Op::Mul => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_mul(src[lane]);
}
}
Op::MulHi => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = mulhi32(r[d][lane], src[lane]);
}
}
Op::Xor => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] ^= src[lane];
}
}
Op::Or => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] |= src[lane];
}
}
Op::Rotl => {
let n = ins.rot;
for lane in 0..LANES {
r[d][lane] = r[d][lane].rotate_left(n);
}
}
Op::Rotr => {
let src = r[a];
for lane in 0..LANES {
r[d][lane] = r[d][lane].rotate_right(src[lane] & 31);
}
}
Op::Mad => {
let src = r[a];
let src2 = r[ins.src2 as usize];
for lane in 0..LANES {
r[d][lane] = src[lane].wrapping_mul(src2[lane]).wrapping_add(r[d][lane]);
}
}
Op::Shfl => {
let src = r[a];
let m = ins.mask as usize;
for lane in 0..LANES {
r[d][lane] ^= src[lane ^ m];
}
}
Op::Load if ins.width == 1 => {
for lane in 0..LANES {
idx[lane] = load_index(era, ins, r[a][lane], mask, log2);
}
*items_derived += ds.fetch(idx, val, layout);
for lane in 0..LANES {
r[d][lane] ^= val[lane];
}
}
Op::Load => {
// Read-width experiment: `width` words from the aligned address, every word folded into dst.
let width = ins.width as usize;
let align = !(ins.width as u32 - 1);
for lane in 0..LANES {
idx[lane] = load_index(era, ins, r[a][lane], mask, log2) & align;
}
let mut vals = [[0u32; 16]; LANES];
*items_derived += ds.fetch_wide(idx, width, &mut vals, layout);
for lane in 0..LANES {
r[d][lane] = fold_words(r[d][lane], &vals[lane][..width]);
}
}
Op::Scratch => {
// handled by the caller (interpret_warp_init), which owns the unit's scratch model
}
Op::Hot => {
// Hot-table experiment: one word of the epoch table at the multiply-shift index, plain xor fold.
let h = ds.hot.as_ref().expect("a hot load needs the hot table");
let n = h.n_words();
for lane in 0..LANES {
r[d][lane] ^= h.at(hot_index(r[a][lane], n));
}
}
Op::WLoad => {
// Lane 0's register, masked, aligned down to 32 words; lane l reads word base + l.
let base = (r[a][0] & mask) & !31;
for lane in 0..LANES {
idx[lane] = base + lane as u32;
}
*items_derived += ds.fetch(idx, val, Layout::LINEAR);
for lane in 0..LANES {
r[d][lane] ^= val[lane];
}
}
}
}
/// The 32 hashes of one warp (`cpuWarp` in the Swift).
pub fn hash_warp(program: &Program, base_nonce: u32, ds: &DatasetSource) -> [u64; LANES] {
interpret_warp(program, base_nonce, ds).hashes
}
/// Everything a node needs to verify blocks of one epoch on one day: the program for the epoch seed and the
/// dataset source for the day key. Building one in memory-hard mode fills the 256 MiB cache (about 0.2 s on
/// one core); keep it for the whole epoch and share it between threads (`&Epoch` is `Send + Sync`).
pub struct Epoch {
pub program: Program,
pub dataset: DatasetSource,
}
/// Default dataset size: 2^28 words = 1 GiB.
pub const DEFAULT_DATASET_LOG2: u32 = 28;
/// Days a day index lies after the network's genesis day (0 for the genesis day and any day before it). The node's
/// entry; the same function as `memhard::days_since_genesis`.
pub fn days_since_genesis(day_index: u64, genesis_day_index: u64) -> u64 {
crate::memhard::days_since_genesis(day_index, genesis_day_index)
}
impl Epoch {
pub fn new(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32) -> Self {
Self { program: generate(seed), dataset: DatasetSource::new(day, mode, dataset_log2) }
}
/// [`Epoch::new`] with a load class (read-width experiment; Counter ASIC 2.0: the class's mixer multiplier
/// shapes the dataset, the cache is the genesis size since a string day has no day index).
pub fn new_class(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32, class: LoadClass) -> Self {
Self::new_class_day(seed, day, mode, dataset_log2, class, 0)
}
/// [`Epoch::new_class`] on day `days_since_genesis` of the growth schedule (the cache of
/// `memhard::cache_log2_words` for a class with the growth rule; the dataset size is the caller's).
pub fn new_class_day(seed: &str, day: &str, mode: DatasetMode, dataset_log2: u32, class: LoadClass, days_since_genesis: u64) -> Self {
let shape = Shape::for_class_day(&class, days_since_genesis);
let program = generate_class(seed, class);
let mut dataset = DatasetSource::new_shape(day, mode, dataset_log2, shape);
// hot-table experiment: a hot class fills its table from the seed bytes
dataset.attach_hot_for(&program);
Self { program, dataset }
}
/// `dataset[w]` as this epoch's program reads it: under the program's layout (era layout; linear for v2).
pub fn dataset_word(&self, w: u32) -> u32 {
self.dataset.word_at(self.program.class.layout(), w)
}
/// The production shape: memory-hard, 1 GiB dataset.
pub fn memory_hard(seed: &str, day: &str) -> Self {
Self::new(seed, day, DatasetMode::MemoryHard, DEFAULT_DATASET_LOG2)
}
/// The chain's shape: program from the 32-byte epoch seed (devnet: the epoch block hash; later the VDF
/// output) through the version 2 generator and its acceptance rule, and the cache from
/// `seed_words_from_bytes(day_bytes)` (`bind::day_bytes`). Memory-hard, 1 GiB dataset. `label` is only
/// recorded in emitted packs.
pub fn from_seed_bytes(epoch_seed: &[u8], day_bytes: &[u8], label: &str) -> Self {
Self::from_seed_bytes_class(epoch_seed, day_bytes, label, LoadClass::V2)
}
/// [`Epoch::from_seed_bytes`] with a load class (read-width experiment; Counter ASIC 2.0: the class's mixer
/// multiplier shapes the dataset). Day 0 of the growth schedule: the 2^26-word cache and the 2^28-word dataset,
/// which is every devnet pack and vector. A node past the first doubling calls [`Epoch::from_seed_bytes_day`].
pub fn from_seed_bytes_class(epoch_seed: &[u8], day_bytes: &[u8], label: &str, class: LoadClass) -> Self {
Self::from_seed_bytes_day(epoch_seed, day_bytes, label, class, 0, DEFAULT_DATASET_LOG2)
}
/// The chain's shape on day `days_since_genesis` (`memhard::days_since_genesis(day_index(header), day_index(genesis))`,
/// the node's two day indices): the program of the class, and under the class's growth rule the cache of
/// `memhard::cache_log2_words(d)` and the dataset of `memhard::dataset_log2_words(genesis_dataset_log2, d)`
/// (the genesis size is 28 for the 1 GiB devnet, 29 for the designed 2 GiB). Without the growth rule the cache
/// is 2^26 words and the dataset `2^genesis_dataset_log2` on every day.
pub fn from_seed_bytes_day(epoch_seed: &[u8], day_bytes: &[u8], label: &str, class: LoadClass, days_since_genesis: u64, genesis_dataset_log2: u32) -> Self {
let program = crate::generator::generate_from_seed_bytes_class(label, epoch_seed, class);
let key = crate::seed::seed_words_from_bytes(day_bytes);
let shape = Shape::for_class_day(&class, days_since_genesis);
let dataset_log2 = if class.growth { crate::memhard::dataset_log2_words(genesis_dataset_log2, days_since_genesis) } else { genesis_dataset_log2 };
let mut dataset = DatasetSource::from_key_shape(key, DatasetMode::MemoryHard, dataset_log2, shape);
dataset.key_bytes = day_bytes.to_vec();
dataset.attach_hot_for(&program);
Self { program, dataset }
}
/// The chain's shape with the program class (Counter ASIC 2.0, 5 October 2026): what the node's engine and the
/// miner's pack export build from the seeds a block template carries. Class v2 is [`Epoch::from_seed_bytes`]
/// exactly (the era bytes are ignored and not recorded); class v3 draws from [`crate::generator::V3_CLASS`]
/// with generator version 3 and records the era seed bytes (`E_n`) in the program for the pack.
pub fn from_chain_seeds(epoch_seed: &[u8], day_bytes: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, label: &str) -> Self {
Self { program: Self::chain_program(epoch_seed, era_bytes, class, label), dataset: Self::chain_dataset(day_bytes, class) }
}
/// The program alone of [`Epoch::from_chain_seeds`] (no cache fill): for an engine that shares the day's cache.
pub fn chain_program(epoch_seed: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, label: &str) -> Program {
crate::generator::generate_from_seed_bytes_program_class(label, epoch_seed, class, era_bytes)
}
/// [`Epoch::chain_program`] at a rung of the latency ladder (`docs/design/latency-ladder.md`): `shadow_reps` is
/// the shadow pass count the chain's step gives the epoch (0 = the class's own, which is [`Epoch::chain_program`]
/// byte for byte). The node's engine and the miner's pack export call this with the step the template names.
pub fn chain_program_shadow(epoch_seed: &[u8], era_bytes: Option<&[u8]>, class: ProgramClass, shadow_reps: u16, label: &str) -> Program {
crate::generator::generate_from_seed_bytes_program_class_shadow(label, epoch_seed, class, era_bytes, shadow_reps)
}
/// The day's cache and dataset of [`Epoch::from_chain_seeds`], the one entry the node's engine builds a day
/// cache through. The class is an argument because the Counter ASIC 2.0 integration gives class v3 its own item
/// construction (the mixer multiplier) and cache size schedule (ca2-mixer); today both classes build the day of
/// [`Epoch::from_seed_bytes`], and the engine keys its day caches on `(day, class)` so the two never share one.
pub fn chain_dataset(day_bytes: &[u8], class: ProgramClass) -> DatasetSource {
Self::chain_dataset_day(day_bytes, class, 0, DEFAULT_DATASET_LOG2)
}
/// [`Epoch::chain_dataset`] with the day's position since genesis and the network's genesis dataset size: the
/// entry the node's engine and the miner's export build every day cache through, so the cache growth schedule
/// of spec 01 section 1.13.3 has one place to act (ca2-mixer, 5 October 2026, `docs/plans/mixer-x4.md`): the
/// class's load class gives the mixer multiplier and whether the growth rule applies (`Shape::for_class_day`);
/// under the rule the cache is `2^memhard::cache_log2_words(d)` words and the dataset
/// `2^memhard::dataset_log2_words(genesis_dataset_log2, d)`; without it (class v2) the cache is 2^26 words and
/// the dataset the genesis size on every day. `days_since_genesis` is [`days_since_genesis`] of the block's and
/// the genesis header's day indices.
pub fn chain_dataset_day(day_bytes: &[u8], class: ProgramClass, days_since_genesis: u64, genesis_dataset_log2: u32) -> DatasetSource {
let lc = class.load_class();
let shape = Shape::for_class_day(&lc, days_since_genesis);
let dataset_log2 = if lc.growth { crate::memhard::dataset_log2_words(genesis_dataset_log2, days_since_genesis) } else { genesis_dataset_log2 };
let key = crate::seed::seed_words_from_bytes(day_bytes);
let mut dataset = DatasetSource::from_key_shape(key, DatasetMode::MemoryHard, dataset_log2, shape);
dataset.key_bytes = day_bytes.to_vec();
dataset
}
/// The 32 hashes of the warp starting at `base_nonce`.
pub fn hash_warp(&self, base_nonce: u32) -> [u64; LANES] {
hash_warp(&self.program, base_nonce, &self.dataset)
}
pub fn interpret_warp(&self, base_nonce: u32) -> WarpResult {
interpret_warp(&self.program, base_nonce, &self.dataset)
}
/// The hash of one nonce. The verification unit is a warp, so the 31 sibling nonces of the aligned
/// 32-nonce group are computed too (the shuffles couple the lanes).
pub fn hash(&self, nonce: u32) -> u64 {
self.hash_warp(nonce & !31)[(nonce & 31) as usize]
}
/// `hash(nonce) <= target`. The hash is 64 bits; the fork maps it into its 256-bit target space.
pub fn verify_block(&self, nonce: u32, target: u64) -> bool {
self.hash(nonce) <= target
}
}
/// One-shot `verify_block(seed, nonce, target)`: builds the epoch (cache fill included) and checks. For a
/// node use [`Epoch`] and keep it; this exists for scripts and tests.
pub fn verify_block(seed: &str, day: &str, nonce: u32, target: u64) -> bool {
Epoch::memory_hard(seed, day).verify_block(nonce, target)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn closed_form_head_matches_pack() {
// vectors.json dataset_head for igneum-genesis (closed form), day 2026-10-03.
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28);
assert_eq!(ds.word(0), 0x82174c0f);
assert_eq!(ds.word(1), 0x577bdb9c);
assert_eq!(ds.word(0x0fffffff), 0xf78c84a4);
}
/// Read-width experiment: the fold for one word is a plain xor; a wide fetch hands each lane the words the
/// scalar path would; two distinct lines give two distinct maps of dst (one point suffices as a smoke check).
#[test]
fn fold_and_wide_fetch() {
assert_eq!(fold_words(0x1234_5678, &[0xdead_beef]), 0x1234_5678 ^ 0xdead_beef);
let w = [1u32, 2, 3, 4];
let x = fold_words(7, &w);
let mut y: u32 = 7 ^ 1;
for &v in &w[1..] {
y = y.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ v;
}
assert_eq!(x, y);
assert_ne!(fold_words(7, &[1, 2, 3, 4]), fold_words(7, &[1, 2, 3, 5]));
let ds = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 20);
let mut base = [0u32; LANES];
for (k, b) in base.iter_mut().enumerate() {
*b = ((k as u32).wrapping_mul(0x9E37_79B1) & ds.mask) & !15;
}
let mut out = [[0u32; 16]; LANES];
let items = ds.fetch_wide(&base, 16, &mut out, Layout::LINEAR);
assert!(items >= 1 && items <= LANES);
for k in 0..LANES {
for j in 0..16 {
assert_eq!(out[k][j], ds.word(base[k] + j as u32), "lane {k} word {j}");
}
}
let mut base4 = base;
for b in base4.iter_mut() {
*b += 8;
}
let items4 = ds.fetch_wide(&base4, 4, &mut out, Layout::LINEAR);
assert_eq!(items4, items);
for k in 0..LANES {
for j in 0..4 {
assert_eq!(out[k][j], ds.word(base4[k] + j as u32));
}
}
}
/// A wide-load program interprets identically on the closed form and through the memory-hard path's fold
/// (the same fold code), and a mixed-class epoch builds and hashes.
/// Variant 5: a fill word is deterministic, a rewrite changes the slot, and a second read of a written slot
/// returns the rewrite, not the fill.
#[test]
fn scratch_model() {
let seed = [1u32, 2, 3, 4, 5, 6, 7, 8];
assert_eq!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 1));
assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2));
assert_ne!(scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 64, 3, 100, 1));
let mut m = ScratchModel::new(256);
let w = [scratch_fill(&seed, 32, 3, 100, 0), scratch_fill(&seed, 32, 3, 100, 1), scratch_fill(&seed, 32, 3, 100, 2)];
let x = m.rmw(&seed, 32, 3, 100, 0xabcd);
assert_eq!(x, fold_words(0xabcd, &w));
let x2 = m.rmw(&seed, 32, 3, 100, 0xabcd);
assert_eq!(x2, fold_words(0xabcd, &scratch_rewrite(x, &w)));
assert_eq!(m.reads, 2);
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::scratch(4, 128));
assert_eq!(e.program.scratch_ops_per_hash(), 32);
assert_eq!(e.hash_warp(0), e.hash_warp(0));
}
/// Era layout: the load address stays inside the site's window and below the mask at every dataset size (the
/// window floor of 2^26 words clamps the shrink), the interleaved memory-hard dataset reads item(t(w))[j(w)] and
/// is the same prefix at 2^20 and 2^22 words, the wide fetch agrees word for word, and an era epoch hashes
/// deterministically through the interpreter and the single-nonce API.
#[test]
fn era_windows_layout_and_epochs() {
let eb = EraParams::test_era_bytes("igneum-era-test/1");
let c = LoadClass::era(LoadClass::V2, &eb, &[1]);
let e = c.era.unwrap();
let mut ins = Instr { op: Op::Load, dst: 0, src: 1, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 2, off: 3 };
let mut s = crate::seed::SplitMix64::new(7);
for log2 in [20u32, 26, 27, 28, 29] {
let mask = (1u64 << log2) as u32 - 1;
let k = ins.win.min(log2.saturating_sub(26) as u8) as u32;
for _ in 0..1000 {
let x = s.next() as u32;
let idx = load_index(Some(&e), &ins, x, mask, log2);
assert!(idx <= mask);
let (wm, off) = window(&ins, mask, log2);
assert_eq!(idx & !wm, off, "log2 {log2}");
assert_eq!(wm, mask >> k);
assert_eq!(idx, ((x.wrapping_mul(e.stride_mul).rotate_left(e.stride_rot) & wm) | off) & mask);
}
}
ins.win = 0;
assert_eq!(load_index(None, &ins, 0xdead_beef, 0x0fff_ffff, 28), 0xdead_beef & 0x0fff_ffff);
// the interleaved dataset: one day cache, the layout per program
let l = e.layout();
assert_eq!(l.pos, [1, 3, 8, 13]);
let small = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 20);
let big = DatasetSource::new("2026-10-03", DatasetMode::MemoryHard, 22);
let m = small.memhard().unwrap();
for w in [0u32, 1, 4, 5, 255, 256, 4095, 8192, 0x0f_ffff] {
let (t, j) = l.split(w);
assert_eq!(small.word_at(l, w), crate::memhard::derive_item(t, &m.params, &m.cache)[j as usize], "w {w}");
assert_eq!(small.word_at(l, w), big.word_at(l, w), "prefix at w {w}");
assert_eq!(small.word(w), small.word_at(Layout::LINEAR, w));
}
let mut idx = [0u32; LANES];
for (k, i) in idx.iter_mut().enumerate() {
*i = (k as u32).wrapping_mul(0x9E37_79B1) & small.mask;
}
let mut out = [0u32; LANES];
small.fetch(&idx, &mut out, l);
for k in 0..LANES {
assert_eq!(out[k], small.word_at(l, idx[k]));
}
// a 16-byte era: the wide fetch keeps a lane's four words in one item
let eb3 = EraParams::test_era_bytes("igneum-era-test/3");
let c3 = LoadClass::era(LoadClass::fixed(4, 16), &eb3, &[4]);
let l3 = c3.layout();
assert_eq!(l3.pos[..2], [0, 1]);
let mut base = [0u32; LANES];
for (k, b) in base.iter_mut().enumerate() {
*b = ((k as u32).wrapping_mul(0x9E37_79B1) & small.mask) & !3;
}
let mut wide = [[0u32; 16]; LANES];
small.fetch_wide(&base, 4, &mut wide, l3);
for k in 0..LANES {
for j in 0..4 {
assert_eq!(wide[k][j], small.word_at(l3, base[k] + j as u32), "lane {k} word {j}");
}
}
// era epochs hash deterministically, differ per era, and the single-nonce API agrees with the warp
let mut seen = std::collections::HashSet::new();
for (n, c) in [(1u64, c), (3, c3)] {
let ep = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::MemoryHard, 20, c);
assert_eq!(ep.dataset_word(5), ep.dataset.word_at(c.layout(), 5));
let a = ep.hash_warp(64);
assert_eq!(a, ep.hash_warp(64));
assert_eq!(ep.hash(64 + 5), a[5]);
assert!(seen.insert(a[0]), "era {n}");
let closed = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, c);
assert_ne!(closed.hash_warp(64), a);
}
}
/// Hot-table experiment: an epoch of a hot class carries the table, hashes deterministically and differs from
/// version 2; the reference interpreter agrees with a hand-stepped hot load; a hot program without its table is
/// refused.
#[test]
fn hot_epochs_hash() {
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(32, 4));
let h = e.dataset.hot.as_ref().expect("the epoch fills the hot table");
assert_eq!(h.n_words(), 1 << 23);
assert_eq!(h.key, crate::memhard::hot_key(b"igneum-genesis"));
let v2 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::V2);
let a = e.hash_warp(0);
assert_eq!(a, e.hash_warp(0));
assert_ne!(a, v2.hash_warp(0));
assert_ne!(a[0], a[1]);
// the same program under a 64 MiB table reads other words
let e64 = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, LoadClass::hot(64, 4));
assert_eq!(e64.program.instrs, e.program.instrs);
assert_ne!(e64.hash_warp(0), a);
// from seed bytes, the chain's shape, with a hot class
let genesis = crate::bind::unhex("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07").unwrap();
let ec = Epoch::from_seed_bytes_class(&genesis, &crate::bind::day_bytes(20_730), "devnet", LoadClass::hot(32, 2));
assert_eq!(ec.dataset.hot.as_ref().unwrap().key, crate::memhard::hot_key(&genesis));
assert_eq!(ec.hash_warp(0), ec.hash_warp(0));
}
#[test]
#[should_panic(expected = "needs the epoch's hot table")]
fn hot_program_without_a_table_is_refused() {
let p = generate_class("igneum-genesis", LoadClass::hot(32, 4));
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 20);
let _ = hash_warp(&p, 0, &ds);
}
#[test]
fn wide_class_epochs_hash() {
for name in ["w16", "w64x4", "50,35,15"] {
let c = LoadClass::parse(name).unwrap();
let e = Epoch::new_class("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 20, c);
assert_eq!(e.program.class, c);
let a = e.hash_warp(0);
let b = e.hash_warp(0);
assert_eq!(a, b);
assert_ne!(a[0], a[1]);
}
}
#[test]
fn closed_form_genesis_vector_lane0() {
// Generator v2 vectors (4 October 2026), proto-cuda/packs/igneum-genesis/vectors.json.
let e = Epoch::new("igneum-genesis", "2026-10-03", DatasetMode::ClosedForm, 28);
let w = e.hash_warp(0);
assert_eq!(w[0], 0x31e7555c3dfd007f);
assert_eq!(w[31], 0xaab617183923ab2a);
assert_eq!(e.hash(0), w[0]);
assert_eq!(e.hash(31), w[31]);
assert!(e.verify_block(0, u64::MAX));
assert!(e.verify_block(0, w[0]));
assert!(!e.verify_block(0, w[0] - 1));
}
}

View file

@ -0,0 +1,275 @@
//! The per-day item-derivation program (Counter ASIC 3.0 item 2, `docs/plans/counter-asic-3-derivation.md`): the
//! soundness runs on the CPU.
//!
//! 1. By hand: the derived item of class dr736 restated with the scalar reference on a small cache equals the
//! verifier's batched (SoA) derivation, at the item boundary and the index wrap.
//! 2. The v2 and v3 paths are untouched: the derivation field is 0 on both, their items are the fixed-mixer items.
//! 3. Determinism and the stream: two epochs agree on every vector and file; the program is the continuation of the
//! mixer's stream (the first draw of the program follows the 40 mixer draws); a different day draws a different
//! program; the class is in the program id and the name.
//! 4. Stats beside x8: bit balance and single-bit avalanche of the derived items and of the hash, on the same seeds.
//! 5. A program whose text is the kernels': every instruction's C text evaluated by hand on one state matches the
//! scalar reference (the text forms are what Metal, CUDA and OpenCL compile).
use igneum_pow::derive::{instr_text, run_round_scalar, DOp, DeriveProgram, DERIVE_LEN_X8, DERIVE_PROGRAMS};
use igneum_pow::emit::export_pack;
use igneum_pow::generator::{generate_from_seed_bytes_class, LoadClass, V3_CLASS};
use igneum_pow::memhard::{derive_item, derive_items, mixer, round_key, Cache, MixParams, Shape, ITEM_ROUNDS};
use igneum_pow::seed::{day_key, SplitMix64};
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch};
const DAY: &str = "2026-10-03";
/// The item of a derivation class restated by hand with the scalar reference.
fn item_by_hand(t: u32, mp: &MixParams, cache: &Cache) -> [u32; 16] {
let prog = mp.derive.as_ref().expect("a derivation class");
let mut s = [0u32; 16];
s[..8].copy_from_slice(&mp.key);
for i in 0..8 {
s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]);
}
for r in 0..ITEM_ROUNDS {
run_round_scalar(&prog.rounds[r], &mut s);
let line = cache.line(s[0]);
for i in 0..16 {
s[i] ^= line[i];
}
}
run_round_scalar(&prog.rounds[ITEM_ROUNDS], &mut s);
s
}
#[test]
fn derived_item_by_hand_and_in_batches() {
let key = day_key(DAY);
let cache = Cache::fill_log2(key, 16);
let shape = Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false };
let mp = MixParams::with_shape(key, shape);
let prog = mp.derive.as_ref().unwrap();
assert_eq!(prog.rounds.len(), DERIVE_PROGRAMS);
assert!(prog.check().is_ok());
assert_eq!(shape.derive_instrs_per_item(), 9 * DERIVE_LEN_X8);
assert_eq!(shape.mixers_per_item(), 0);
for t in [0u32, 1, 2, 15, 16, 17, 12_345, (1 << 28) - 1, u32::MAX - 1, u32::MAX] {
assert_eq!(derive_item(t, &mp, &cache), item_by_hand(t, &mp, &cache), "t {t}");
}
// a batch of 32 distinct items against one at a time, and a short batch
let ts: Vec<u32> = (0..32).map(|k| k * 7_919 + 3).collect();
let mut out = [[0u32; 16]; 32];
derive_items(&ts, &mp, &cache, &mut out);
for (k, &t) in ts.iter().enumerate() {
assert_eq!(out[k], item_by_hand(t, &mp, &cache), "batch slot {k}");
}
let mut out5 = [[0u32; 16]; 5];
derive_items(&ts[..5], &mp, &cache, &mut out5);
assert_eq!(&out5[..], &out[..5]);
// the fixed mixer of the same key gives other items
let v3 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 16, derive_len: 0, state: false });
assert!(v3.derive.is_none());
assert_ne!(derive_item(0, &v3, &cache), derive_item(0, &mp, &cache));
}
#[test]
fn v2_and_v3_are_untouched() {
assert_eq!(LoadClass::V2.derive_len, 0);
assert_eq!(V3_CLASS.derive_len, 0);
assert_eq!(LoadClass::MX8.derive_len, 0);
assert_eq!(Shape::V2.derive_len, 0);
assert!(!Shape::for_class(&V3_CLASS).is_derived());
let key = day_key(DAY);
let cache = Cache::fill_log2(key, 16);
// the version 2 item restated by hand (the mixer_mult_by_hand test of memhard.rs, m = 1)
let v2 = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: 0, state: false });
let t = 12_345u32;
let mut s = [0u32; 16];
s[..8].copy_from_slice(&key);
for i in 0..8 {
s[8 + i] = t.wrapping_mul(v2.mul[i]).wrapping_add(v2.rc[i]);
}
for r in 0..8usize {
mixer(&mut s, round_key(r), &v2);
let line = cache.line(s[0]);
for i in 0..16 {
s[i] ^= line[i];
}
}
mixer(&mut s, round_key(8), &v2);
assert_eq!(derive_item(t, &v2, &cache), s);
// the mixer constants of the derivation class are the v2 draws (the stream continues after them)
let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false });
assert_eq!((dr.rot, dr.mul, dr.rc), (v2.rot, v2.mul, v2.rc));
}
#[test]
fn stream_class_name_and_id() {
let key = day_key(DAY);
// the program is the continuation of the mixer stream: 40 draws, then the program
let mut rng = SplitMix64::new(key[0] as u64 | ((key[1] as u64) << 32));
for _ in 0..40 {
rng.next();
}
let expect = DeriveProgram::draw(&mut rng, DERIVE_LEN_X8);
let mp = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8, state: false });
assert_eq!(mp.derive.as_ref().unwrap(), &expect);
// another day, another program; another length, another program
let other = MixParams::with_shape(day_key("2026-10-04"), Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: DERIVE_LEN_X8, state: false });
assert_ne!(other.derive.as_ref().unwrap().fingerprint(), expect.fingerprint());
let short = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 26, derive_len: 368, state: false });
assert_eq!(short.derive.as_ref().unwrap().instr_count(), 9 * 368);
// the class: name, parse, id, and the v2 program stream (v2 loads, no width roll)
let c = LoadClass::DR736;
assert_eq!(c.name(), "dr736");
assert_eq!(LoadClass::parse("dr736"), Some(c));
assert_eq!(LoadClass::parse("dr368"), Some(LoadClass::MX4.with_derive(368)));
assert_eq!(LoadClass::parse("dr0"), None);
assert_eq!(LoadClass::parse("dr5000"), None);
assert!(c.v2_loads() && !c.takes_width_roll() && c.growth && c.mixer_mult == 1 && c.is_derived());
let p = generate_from_seed_bytes_class("igneum-genesis", b"igneum-genesis", c);
let v2 = generate_from_seed_bytes_class("igneum-genesis", b"igneum-genesis", LoadClass::V2);
let mx8 = generate_from_seed_bytes_class("igneum-genesis", b"igneum-genesis", LoadClass::MX8);
assert_eq!(p.instrs, v2.instrs, "the v2 program of the seed");
assert_ne!(p.program_id(), v2.program_id());
assert_ne!(p.program_id(), mx8.program_id());
assert_ne!(
generate_from_seed_bytes_class("igneum-genesis", b"igneum-genesis", LoadClass::MX4.with_derive(368)).program_id(),
p.program_id()
);
}
#[test]
fn determinism_and_pack_text() {
let a = Epoch::new_class("igneum-genesis", DAY, DatasetMode::MemoryHard, 20, LoadClass::DR736);
let b = Epoch::new_class("igneum-genesis", DAY, DatasetMode::MemoryHard, 20, LoadClass::DR736);
let pa = export_pack(&a, DAY, "test");
let pb = export_pack(&b, DAY, "test");
assert_eq!(pa.outs, pb.outs);
assert_eq!(pa.files, pb.files);
let names: Vec<&str> = pa.files.iter().map(|(n, _)| n.as_str()).collect();
assert!(names.contains(&"memhard.h") && names.contains(&"memhard.metal"));
for (name, text) in &pa.files {
if name == "memhard.h" || name == "memhard.metal" || name == "kernel.cl" {
for r in 0..DERIVE_PROGRAMS {
assert!(text.contains(&format!("mh_round_{r}(")), "{name} carries round program {r}");
}
assert!(!text.contains("mh_mixer(s,"), "{name}: no mixer application under a derivation program");
let dp = a.dataset.memhard().unwrap().params.derive.as_ref().unwrap();
// every instruction's text appears, in order, inside the round functions
let first = instr_text(&dp.rounds[0][0]);
assert!(text.contains(&first), "{name} carries the first instruction {first}");
}
if name == "program.h" {
assert!(text.contains("#define IGNEUM_LOAD_CLASS \"dr736\""));
assert!(text.contains("#define IGNEUM_DERIVE_LEN 736"));
assert!(text.contains("#define IGNEUM_CLASS_DERIVE_LEN 736"));
assert!(!text.contains("#define IGNEUM_MIXER_MULT"));
assert!(text.contains("#define IGNEUM_GENERATOR 2"));
}
if name == "program.json" {
assert!(text.contains("\"derive_len\": 736"));
assert!(text.contains("\"programs\": ["));
serde_json::from_str::<serde_json::Value>(text).expect("valid JSON");
}
}
// the vectors are the interpreter's
assert_eq!(pa.outs[0], a.hash_warp(0));
}
/// Bit balance and single-bit avalanche of the derived items (t flipped one bit) and of the hash, beside x8.
#[test]
fn stats_beside_x8() {
let key = day_key(DAY);
let cache = Cache::fill_log2(key, 18);
let dr = MixParams::with_shape(key, Shape { mixer_mult: 1, cache_log2_words: 18, derive_len: DERIVE_LEN_X8, state: false });
let x8 = MixParams::with_shape(key, Shape { mixer_mult: 8, cache_log2_words: 18, derive_len: 0, state: false });
for (label, mp) in [("dr736", &dr), ("x8", &x8)] {
let n = 2048u32;
let mut ones = [0u32; 512];
let mut flips = 0u64;
let mut flip_n = 0u64;
let mut seen = std::collections::HashSet::new();
for t in 0..n {
let a = derive_item(t * 2_654_435_761, mp, &cache);
assert!(seen.insert(a), "{label}: duplicate item");
for i in 0..16 {
for b in 0..32 {
ones[i * 32 + b] += (a[i] >> b) & 1;
}
}
let bit = t % 32;
let c = derive_item((t * 2_654_435_761) ^ (1 << bit), mp, &cache);
for i in 0..16 {
flips += (a[i] ^ c[i]).count_ones() as u64;
}
flip_n += 512;
}
let avalanche = flips as f64 / flip_n as f64 * 100.0;
let worst_z = ones.iter().map(|&o| ((o as f64 - n as f64 / 2.0) / (n as f64 / 4.0).sqrt()).abs()).fold(0.0, f64::max);
println!("{label}: avalanche {avalanche:.2} percent, worst bit z {worst_z:.2}");
assert!((48.5..=51.5).contains(&avalanche), "{label}: avalanche {avalanche}");
assert!(worst_z < 4.5, "{label}: worst bit z {worst_z}");
}
// the hash on the same seeds: avalanche across a nonce flip within the unit
let e = Epoch::new_class("igneum-genesis", DAY, DatasetMode::MemoryHard, 20, LoadClass::DR736);
let w0 = e.hash_warp(0);
let w1 = e.hash_warp(32);
let mut d = 0u64;
for l in 0..32 {
d += (w0[l] ^ w1[l]).count_ones() as u64;
}
let av = d as f64 / (32.0 * 64.0) * 100.0;
println!("hash unit 0 against unit 1: {av:.2} percent of bits differ");
assert!((44.0..=56.0).contains(&av));
}
/// The kernels' text forms: every form's C text, read back into the scalar reference's arithmetic by hand.
#[test]
fn text_forms_match_scalar_reference() {
let mut rng = SplitMix64::new(42);
let p = DeriveProgram::draw_candidate(&mut rng, 64, 0);
let mut s = [0u32; 16];
for (i, v) in s.iter_mut().enumerate() {
*v = 0x9e37_79b9u32.wrapping_mul(i as u32 + 1);
}
let mut seen = [false; 12];
for ins in p.rounds.iter().flatten() {
seen[ins.op as usize] = true;
let before = s;
run_round_scalar(std::slice::from_ref(ins), &mut s);
let (d, c, b) = (ins.dst as usize, ins.src as usize, ins.src2 as usize);
let (dv, cv, bv) = (before[d], before[c], before[b]);
let want = match ins.op {
DOp::Add => dv.wrapping_add(cv),
DOp::Sub => dv.wrapping_sub(cv),
DOp::Xor => dv ^ cv,
DOp::Mul => dv.wrapping_mul(cv | 1),
DOp::Rot => dv.rotate_left(ins.rot as u32).wrapping_add(cv),
DOp::XRot => (dv ^ cv).rotate_left(ins.rot as u32),
DOp::AddC => dv.wrapping_add(cv.wrapping_add(ins.imm)),
DOp::XorC => dv ^ cv ^ ins.imm,
DOp::MulC => (dv ^ cv).wrapping_mul(ins.imm),
DOp::MulC2 => dv.wrapping_mul(ins.imm).wrapping_add(cv),
DOp::AndX => dv ^ (cv & bv),
DOp::OrX => dv.wrapping_add(cv | bv),
};
assert_eq!(s[d], want, "{}", instr_text(ins));
for i in 0..16 {
if i != d {
assert_eq!(s[i], before[i], "only the destination changes: {}", instr_text(ins));
}
}
assert!(instr_text(ins).starts_with(&format!("s[{d}]")));
}
assert!(seen.iter().all(|s| *s), "64 x 9 draws cover every form");
}
/// The dataset source of the class on a day: the verifier's `word` path derives through the program.
#[test]
fn dataset_source_word_path() {
let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape { mixer_mult: 1, cache_log2_words: 16, derive_len: DERIVE_LEN_X8, state: false });
let m = ds.memhard().unwrap();
let item = derive_item(3, &m.params, &m.cache);
for j in 0..16u32 {
assert_eq!(ds.word(3 * 16 + j), item[j as usize]);
}
assert_eq!(ds.word((1 << 20) + 5), ds.word(5), "the mask");
}

View file

@ -0,0 +1,356 @@
//! The class v3 dataset construction (mixer x4, cache growth option C; `docs/plans/mixer-x4.md`): the soundness
//! runs the brief asks for, on the CPU, with the packs for the GPU runs written on request.
//!
//! 1. Fuzz: `IGNEUM_MIXER_FUZZ` (default 200) programs through the seam (`ProgramClass::V3`), the contract on every
//! instruction (the v2 program of the seed, instruction for instruction), 4 units each across the 32-bit range
//! including the wrap, interpreted twice on the CPU; with `IGNEUM_MIXER_PACKS_OUT=<dir>` every program is written
//! as a pack with its 4 bases in vectors.json for `packbench` and the OpenCL host (the Metal fuzz).
//! 2. Stats: bit balance and single-bit-flip avalanche of the v3 hash against v2 on the same programs and nonces.
//! 3. Edge: the dataset at word 0, word MASK and the item boundary, derived through the interpreter's fetch path and
//! by hand at every multiplier 1, 2, 4, 8, on a small cache.
//! 4. Determinism: two independent epochs of the same seed and day agree on every vector and every emitted file.
//!
//! Counter ASIC 3.0 gate run (6 October 2026): `IGNEUM_MIXER_CLASS=<class>` (a `LoadClass::parse` name, for example
//! `mx8+sh256x27`) runs the fuzz, the stats and the determinism test on that class instead of `V3_CLASS`;
//! `IGNEUM_MIXER_ERA=igneum-era-test/<n>` composes that era over the class as the chain composes it inside class v3
//! (`LoadClass::era(class, E_n, &V3_ALLOWED)`, generator 3). The fuzz contract on a shadow class also checks the block:
//! `S` instructions, none a load, every field in range, and the 64 base instructions equal to the class's without the
//! shadow, draw for draw. The edge test is the dataset's alone (the shadow touches no dataset word) and takes no class.
use igneum_pow::emit::{export_pack, vectors_json};
use igneum_pow::generator::{era_generator_of, generate_era, generate_from_seed_bytes, generate_from_seed_bytes_class, generate_from_seed_bytes_program_class, EraParams, LoadClass, Op, Program, ProgramClass, GENERATOR_VERSION_V3, GENERATOR_VERSION_V4, INSTR_COUNT, V3_ALLOWED, V3_CLASS, V4_CLASS, V4_SHADOW_INSTRS};
use igneum_pow::memhard::{derive_item, mixer, round_key, Cache, MixParams, Shape};
use igneum_pow::seed::{day_key, SplitMix64};
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch};
use std::collections::HashMap;
use std::path::PathBuf;
const DAY: &str = "2026-10-03";
/// The class under test (`IGNEUM_MIXER_CLASS`, default `V3_CLASS`) and the era composed over it (`IGNEUM_MIXER_ERA`,
/// `igneum-era-test/<n>`, default none). With an era the class returned is the era class (the name carries `-era<hex>`).
fn class_under_test() -> (LoadClass, Option<[u8; 32]>) {
let class = std::env::var("IGNEUM_MIXER_CLASS").ok().map(|s| LoadClass::parse(&s).expect("a load class")).unwrap_or(V3_CLASS);
let era = std::env::var("IGNEUM_MIXER_ERA").ok().map(|s| {
assert!(s.starts_with("igneum-era-test/"), "IGNEUM_MIXER_ERA is igneum-era-test/<n>");
EraParams::test_era_bytes(&s)
});
match era {
Some(eb) => (LoadClass::era(class, &eb, &V3_ALLOWED), Some(eb)),
None => (class, None),
}
}
/// The program of `seed` under `class` (an era class carries its era bytes): the seam for `V3_CLASS`, `generate_era`
/// for an era class, the plain class generator otherwise.
fn program_of(seed: &str, class: LoadClass, era: Option<[u8; 32]>) -> Program {
match era {
Some(eb) => generate_era(seed, seed.as_bytes(), LoadClass { era: None, ..class }, &eb, &V3_ALLOWED),
None if class == V3_CLASS => generate_from_seed_bytes_program_class(seed, seed.as_bytes(), ProgramClass::V3, None),
None => generate_from_seed_bytes_class(seed, seed.as_bytes(), class),
}
}
fn contract(p: &Program, seed: &str, class: LoadClass, era: Option<[u8; 32]>) {
if class == V3_CLASS {
assert_eq!(p.generator, GENERATOR_VERSION_V3);
} else if era.is_some() {
// the generator of an era class is the class's (ca3-v4-node 7c22d0d): 4 on V4_CLASS, 3 otherwise
assert_eq!(p.generator, era_generator_of(&LoadClass { era: None, ..class }));
}
assert_eq!(p.class, class);
assert_eq!(p.instrs.len(), INSTR_COUNT);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16);
for (k, i) in p.instrs.iter().enumerate() {
assert!(i.src != i.dst, "#{k}: src == dst");
assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot);
assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask);
assert!(i.dst < 8 && i.src < 8 && i.src2 < 8);
assert_eq!(i.width, 1, "#{k}: a v3 load reads one word");
}
assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program");
if era.is_none() {
// no era: the base program is the v2 program of the seed (the mixer and the shadow draw nothing before it)
let v2 = generate_from_seed_bytes(seed, seed.as_bytes());
assert_eq!(p.instrs, v2.instrs, "the v2 program of the seed under the v3 construction");
assert_eq!(p.attempt, v2.attempt);
}
match class.shadow {
None => assert!(p.shadow.is_empty() && !p.has_shadow()),
Some(sh) => {
// the latency-shadow block (Counter ASIC 3.0 item 8): S ALU instructions, no load, every field in range,
// and the base program is the class's without the shadow, draw for draw (the block is drawn after it)
assert_eq!(p.shadow.len(), sh.instrs as usize, "{seed}: shadow block length");
assert!(p.has_shadow());
assert_eq!(p.shadow_instrs_per_hash(), sh.instrs as usize * sh.reps as usize * 8);
for (k, i) in p.shadow.iter().enumerate() {
assert!(!i.op.is_load() && i.op != Op::WLoad, "shadow #{k}: a load");
assert!(i.src != i.dst, "shadow #{k}: src == dst");
assert!((1..=31).contains(&i.rot), "shadow #{k}: rot {}", i.rot);
assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "shadow #{k}: mask {}", i.mask);
assert!(i.dst < 8 && i.src < 8 && i.src2 < 8 && i.bit < 32, "shadow #{k}: register or bit out of range");
assert_eq!((i.width, i.win, i.off), (1, 0, 0), "shadow #{k}: no load fields");
}
let base = program_of(seed, LoadClass { shadow: None, ..class }, era);
let v4_shape = LoadClass { era: None, shadow: None, ..class } == LoadClass { shadow: None, ..V4_CLASS } && sh.instrs == V4_SHADOW_INSTRS;
if v4_shape {
// the amended class v4 (AP-F8-1, sub-version 2): on every draw path a load's source is a register
// fresh by dataflow (a load keeps freshness only from a fresh source; add, sub, xor, mad, shfl from
// either operand; rotates from their operand; or, mul, mulhi never), so its base program is its own
// stream, not class v3's; what holds is the rule itself, checked here on every load site in draw order
let mut fresh = [true; 8];
for (k, i) in p.instrs.iter().enumerate() {
let (d, a) = (i.dst as usize, i.src as usize);
if i.op.is_load() {
assert!(fresh[a], "{seed}: load #{k} reads r{} which is not fresh by dataflow", i.src);
}
fresh[d] = match i.op {
Op::Load | Op::WLoad | Op::Scratch | Op::Hot => fresh[a],
Op::Add | Op::Sub | Op::Xor | Op::Mad | Op::Shfl => fresh[d] || fresh[a],
Op::Rotl | Op::Rotr => fresh[d],
Op::Or | Op::Mul | Op::MulHi => false,
};
}
} else {
assert_eq!(p.instrs, base.instrs, "{seed}: the base program is the class's without the shadow");
assert_eq!(p.attempt, base.attempt);
}
if era.is_none() || p.generator == GENERATOR_VERSION_V4 {
// a generator-2 program carries the shadow in its id bytes; a class v4 program is generator 4
// (ca3-v4-node 7c22d0d), so its id differs from the generator-3 id of the same seeds
assert_ne!(p.program_id(), base.program_id(), "{seed}: the shadow is in the program id");
} else {
// a generator-3 program's id is program_id(3, seed, attempt), class-independent by construction
// (generator.rs program_id): under the chain's path a class v4 program shares its id with the class
// v3 program of the same seeds until the v4 seam gives it its own generator or puts the class in the
// id (Counter ASIC 3.0 gate run, 6 October 2026: an item for gates G4 and G6, not a hash fault)
assert_eq!(p.generator, GENERATOR_VERSION_V3);
if p.program_id() == base.program_id() {
println!("{seed}: generator-3 program id {:016x} is the same with and without the shadow (the v4 seam item)", p.program_id());
}
}
}
}
}
/// Write a pack whose vectors.json carries `bases` instead of the three standard bases (packbench and the OpenCL
/// host check every unit standalone and the ones inside the batch window).
fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) {
let mut pack = export_pack(e, day, source);
let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect();
let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true);
for f in pack.files.iter_mut() {
if f.0 == "vectors.json" {
f.1 = vj.clone();
}
}
pack.write_to(dir).unwrap();
}
#[test]
fn fuzz_v3_programs_cpu() {
let n: usize = std::env::var("IGNEUM_MIXER_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200);
// IGNEUM_FUZZ_SEED_BASE (default 0) offsets the seed index so a continuous fuzzer (the box's capacity layer,
// infra/build-server/capacity) walks fresh programs round after round; the default run is unchanged
let base: usize = std::env::var("IGNEUM_FUZZ_SEED_BASE").ok().and_then(|s| s.parse().ok()).unwrap_or(0);
let out = std::env::var("IGNEUM_MIXER_PACKS_OUT").ok().map(PathBuf::from);
// IGNEUM_MIXER_CLASS=mx8 fuzzes the x8 candidate as a load class (generator 2 with the class in the id); the
// default is V3_CLASS through the seam; IGNEUM_MIXER_ERA composes a test era over the class (class_under_test)
let (class, era) = class_under_test();
let mut rng = SplitMix64::new(0x6967_6e65_756d_2d6d); // "igneum-m"
let shape = Shape::for_class(&class);
assert_eq!(shape.cache_log2_words, 26);
assert!(shape.mixer_mult > 1);
// one memory-hard source per dataset size (the 256 MiB cache fill is 0.2 s each)
let mut mh: HashMap<u32, DatasetSource> = HashMap::new();
let mut manifest = String::from("pack\tlog2\tprogram_id\tbases\n");
let mut units = 0usize;
let mut wraps = 0usize;
for i in base..base + n {
let seed = format!("igneum-mixer-fuzz/{i}");
let p = program_of(&seed, class, era);
contract(&p, &seed, class, era);
let b0 = (rng.below(8) as u32) * 32;
let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32);
let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32);
let b3 = (rng.next() as u32) & !31;
let bases = [b0, b1, b2, b3];
wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count();
let log2 = [24u32, 26, 28][rng.below(3) as usize];
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, log2, shape));
let e = Epoch { program: p, dataset: ds };
for &b in &bases {
let r1 = e.interpret_warp(b);
let r2 = e.interpret_warp(b);
assert_eq!(r1.hashes, r2.hashes);
assert!(r1.items_derived >= 120 * 32 / 32 && r1.items_derived <= 4_096, "{seed}: {} items", r1.items_derived);
units += 1;
}
if let Some(dir) = &out {
let pack_name = format!("fuzz-{i:03}-{}-l{log2}", class.name());
write_pack_with_bases(&dir.join(&pack_name), &e, DAY, &bases, "igneum-pow tests/mixer.rs fuzz");
manifest.push_str(&format!(
"{pack_name}\t{log2}\t{:016x}\t{}\n",
e.program.program_id(),
bases.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
));
}
mh.insert(log2, e.dataset);
}
println!("fuzz: {n} {} programs, {units} units on the CPU, {wraps} units in the top 256 nonces, seeds {base}..{}", class.name(), base + n);
assert_eq!(units, 4 * n);
assert_eq!(wraps, n);
if let Some(dir) = &out {
std::fs::create_dir_all(dir).unwrap();
std::fs::write(dir.join("manifest.tsv"), manifest).unwrap();
println!("packs written to {}", dir.display());
}
}
/// Bit balance and avalanche of the v3 hash beside v2 on the same program (the TESTS.md section 3 shape, on the
/// CPU, 2^13 nonces per seed): every output bit within 5 sigma of half ones; a single nonce-bit flip moves 50 percent
/// of the output bits within 2 points; no duplicate among the outputs.
#[test]
fn stats_v3_against_v2() {
let n_warps = 256usize; // 8,192 nonces
let (class, era) = class_under_test();
let class_name = if class == V3_CLASS { "v3".to_string() } else { class.name() };
for seed in ["igneum-genesis", "igneum-genesis/stats1"] {
let v3 = Epoch {
program: program_of(seed, class, era),
dataset: DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 24, Shape::for_class(&class)),
};
let v2 = Epoch::new(seed, DAY, DatasetMode::MemoryHard, 24);
for (name, e) in [(class_name.as_str(), &v3), ("v2", &v2)] {
let mut ones = [0u64; 64];
let mut outs = Vec::with_capacity(n_warps * 32);
for w in 0..n_warps {
let h = e.hash_warp(w as u32 * 32);
for &x in &h {
outs.push(x);
for b in 0..64 {
ones[b] += (x >> b) & 1;
}
}
}
let total = (n_warps * 32) as f64;
let sigma = (total / 4.0).sqrt();
for (b, &c) in ones.iter().enumerate() {
let z = (c as f64 - total / 2.0).abs() / sigma;
assert!(z < 5.0, "{seed} {name}: bit {b} ones {c} of {total}, z {z:.2}");
}
// avalanche: flip one bit of the nonce within the unit (lanes 0..31 differ in the low 5 bits) and across
// units (bit 5 and up): compare lane l of unit u with lane l ^ (1 << k) and with unit u ^ (1 << k)
let mut flips = 0u64;
let mut moved = 0u64;
for w in 0..64usize {
let h = e.hash_warp(w as u32 * 32);
for k in 0..5 {
for l in 0..32usize {
moved += (h[l] ^ h[l ^ (1 << k)]).count_ones() as u64;
flips += 1;
}
}
let h2 = e.hash_warp((w ^ 1) as u32 * 32);
for l in 0..32usize {
moved += (h[l] ^ h2[l]).count_ones() as u64;
flips += 1;
}
}
let avg = moved as f64 / flips as f64 / 64.0 * 100.0;
assert!((avg - 50.0).abs() < 2.0, "{seed} {name}: avalanche {avg:.2} percent");
outs.sort_unstable();
let dups = outs.windows(2).filter(|p| p[0] == p[1]).count();
assert_eq!(dups, 0, "{seed} {name}: duplicate outputs");
println!("{seed} {name}: {} outputs, avalanche {avg:.2} percent, worst bit z {:.2}", outs.len(), ones.iter().map(|&c| (c as f64 - total / 2.0).abs() / sigma).fold(0.0, f64::max));
}
assert_ne!(v3.hash_warp(0), v2.hash_warp(0));
}
}
/// The dataset edges under every multiplier on a small cache: word 0, word MASK, the last word of item 0 and the
/// first of item 1, through `DatasetSource::word` and by hand.
#[test]
fn edge_items_every_multiplier() {
let key = day_key(DAY);
let cache = Cache::fill_log2(key, 14);
for m in [1u32, 2, 4, 8] {
let mp = MixParams::with_shape(key, Shape { mixer_mult: m, cache_log2_words: 14, derive_len: 0, state: false });
let by_hand = |t: u32| -> [u32; 16] {
let mut s = [0u32; 16];
s[..8].copy_from_slice(&key);
for i in 0..8 {
s[8 + i] = t.wrapping_mul(mp.mul[i]).wrapping_add(mp.rc[i]);
}
for r in 0..8usize {
for j in 0..m as usize {
mixer(&mut s, round_key(r * m as usize + j), &mp);
}
let line = cache.line(s[0]);
for i in 0..16 {
s[i] ^= line[i];
}
}
for j in 0..m as usize {
mixer(&mut s, round_key(8 * m as usize + j), &mp);
}
s
};
for t in [0u32, 1, 0x0fff_ffff, 0xffff_ffff] {
assert_eq!(derive_item(t, &mp, &cache), by_hand(t), "m {m} item {t}");
}
}
// the interpreter's fetch path at the genesis cache: words 0, 15, 16 and MASK of a 2^20-word dataset agree with
// the item derivation, under v3
let ds = DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 20, Shape::for_class(&V3_CLASS));
let m = ds.memhard().unwrap();
for w in [0u32, 15, 16, 17, ds.mask - 1, ds.mask] {
assert_eq!(ds.word(w), derive_item(w >> 4, &m.params, &m.cache)[(w & 15) as usize]);
assert_eq!(ds.word(w), m.word(w));
}
// a load at an out-of-range register masks to the dataset: the word at mask + 1 is the word at 0
assert_eq!(ds.word(ds.mask.wrapping_add(1)), ds.word(0));
}
/// Two independent epochs of the same seed and day: every vector and every emitted file identical; the pinned v3
/// pack is what a third export writes.
#[test]
fn determinism_v3() {
let (class, era) = class_under_test();
let build = || Epoch {
program: program_of("igneum-genesis", class, era),
dataset: DatasetSource::new_shape(DAY, DatasetMode::MemoryHard, 28, Shape::for_class(&class)),
};
let a = build();
let b = build();
let pa = export_pack(&a, DAY, "a");
let pb = export_pack(&b, DAY, "a");
assert_eq!(pa.outs, pb.outs);
assert_eq!(pa.vectors, pb.vectors);
assert_eq!(pa.files, pb.files);
for (w, warp) in [(0u32, 0usize), (4096, 1), (1_000_000, 2)] {
assert_eq!(a.hash_warp(w), pa.outs[warp]);
}
// the pinned pack of the class: mx8-genesis for V3_CLASS, packs-ca3-shadow/<block> for a shadow class over mx8
// (igneum-genesis, the same day); a class with no pinned pack skips the on-disk comparison and says so
let pinned = if class == V3_CLASS {
Some("../proto-cuda/packs-ca2-mixer/mx8-genesis".to_string())
} else {
class.name().strip_prefix("mx8+").map(|b| format!("../proto-cuda/packs-ca3-shadow/{b}"))
};
let dir = match pinned.map(|d| PathBuf::from(env!("CARGO_MANIFEST_DIR")).join(d)).filter(|d| d.is_dir()) {
Some(d) => d,
None => {
println!("determinism {}: two builds equal; no pinned pack on disk for this class", class.name());
return;
}
};
println!("determinism {}: two builds equal, against the pinned pack {}", class.name(), dir.display());
for (name, text) in &pa.files {
if name == "vectors.json" || name == "vectors.h" {
continue; // the source string differs ("a" here)
}
let on_disk = std::fs::read_to_string(dir.join(name)).unwrap();
assert_eq!(&on_disk, text, "{name}");
}
}

View file

@ -0,0 +1,999 @@
//! The checked-in packs under proto-cuda/packs/ against this crate (their source since generator version 2,
//! 4 October 2026): every program instruction by instruction from its seed bytes, the generator version, attempt
//! and program id, the dataset and cache self-test words, the 96 hash vectors per pack, and every emitted file
//! byte for byte. A pack that drifts from the emitters, or a generator change that moves a vector, fails here.
//!
//! Packs: igneum-genesis-mh and igneum-devnet-v4-epoch0 (memory-hard; the latter from the devnet genesis hash as
//! the epoch seed and the day bytes of 2026-10-04), igneum-genesis and igneum-hourly (closed-form dataset,
//! interpreter regression only); and, under `proto-cuda/packs-ca2-mixer/`, the class v3 packs mx8-genesis and
//! mx8-devnet-epoch0 (Counter ASIC 2.0, 5 October 2026: generator 3 on `V3_CLASS` = mixer x8 with the cache growth
//! rule, decided 22:05 UTC under the delegated rule; the same seeds and days as the two memory-hard v2 packs, so the
//! v2 program and cache carry over and only the dataset words and the hashes change) and the x4 candidate's packs
//! mx4-genesis and mx4-devnet-epoch0 (generator 2 with the load class in the id, the record of the x4 rows).
use igneum_pow::accept;
use igneum_pow::emit::{
cuda_kernel, cuda_kernel_bound, cuda_memhard_header, export_pack, metal_memhard, metal_memhard_for, metal_program,
metal_program_bound, opencl_kernel, opencl_kernel_bound, program_header, program_json, LoadSource,
};
use igneum_pow::generator::{generate_from_seed_bytes, generate_from_seed_bytes_class, generate_from_seed_bytes_program_class, LoadClass, Op, ProgramClass, GENERATOR_VERSION, GENERATOR_VERSION_V3, LOAD_SLOTS, V3_CLASS};
use igneum_pow::memhard::{Shape, CACHE_WORDS};
use igneum_pow::verify::{DatasetMode, DatasetSource, Epoch};
use serde_json::Value;
use std::path::PathBuf;
use std::sync::OnceLock;
const PACKS: [&str; 8] = [
"igneum-genesis-mh",
"igneum-devnet-v4-epoch0",
"igneum-genesis",
"igneum-hourly",
"mx8-genesis",
"mx8-devnet-epoch0",
"mx4-genesis",
"mx4-devnet-epoch0",
];
/// The class v3 packs (generator 3 through the seam).
const PACKS_V3: [&str; 2] = ["mx8-genesis", "mx8-devnet-epoch0"];
fn packs_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs")
}
/// The directory a pack lives in: the class v3 packs under packs-ca2-mixer, the rest under packs.
fn pack_dir(pack: &str) -> PathBuf {
if pack.starts_with("mx4-") || pack.starts_with("mx8-") {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-mixer").join(pack)
} else {
packs_dir().join(pack)
}
}
fn read(pack: &str, file: &str) -> String {
let p = pack_dir(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
fn json(pack: &str, file: &str) -> Value {
serde_json::from_str(&read(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"))
}
fn hex32(v: &Value) -> u32 {
u32::from_str_radix(v.as_str().unwrap().trim_start_matches("0x"), 16).unwrap()
}
fn hex64(v: &Value) -> u64 {
u64::from_str_radix(v.as_str().unwrap().trim_start_matches("0x"), 16).unwrap()
}
fn unhex(v: &Value) -> Vec<u8> {
igneum_pow::bind::unhex(v.as_str().unwrap()).unwrap()
}
/// The epoch a pack describes, rebuilt from program.json alone: the program from `seed_bytes`, the dataset from
/// `dataset.day_bytes` in the pack's mode and size. Memory-hard packs fill a 256 MiB cache (about 0.2 s), so
/// each is built once.
fn epoch(pack: &str) -> &'static Epoch {
static E: OnceLock<Vec<(String, Epoch)>> = OnceLock::new();
let all = E.get_or_init(|| {
PACKS
.iter()
.map(|p| {
let j = json(p, "program.json");
let seed = j["seed"].as_str().unwrap();
let seed_bytes = unhex(&j["seed_bytes"]);
let day_bytes = unhex(&j["dataset"]["day_bytes"]);
let mode = match j["dataset_mode"].as_str().unwrap() {
"memory-hard" => DatasetMode::MemoryHard,
_ => DatasetMode::ClosedForm,
};
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
// a class v3 pack: generator 3 on V3_CLASS through the seam, the era bytes it records, the dataset
// in the class's shape on day 0 (the growth rule's genesis cache: every pinned pack is a day-0 size)
let program = match j["generator"].as_u64().unwrap() as u32 {
GENERATOR_VERSION_V3 => {
let era = j.get("era_seed_bytes").map(unhex);
generate_from_seed_bytes_program_class(seed, &seed_bytes, ProgramClass::V3, era.as_deref())
}
// a generator 2 pack with a load class (the x4 candidate's packs): the class from program.json
_ => match j.get("load_class").and_then(|c| c.as_str()) {
Some(c) => generate_from_seed_bytes_class(seed, &seed_bytes, LoadClass::parse(c).expect("a load class name")),
None => generate_from_seed_bytes(seed, &seed_bytes),
},
};
let shape = Shape::for_class(&program.class);
let mut dataset =
DatasetSource::from_key_shape(igneum_pow::seed::seed_words_from_bytes(&day_bytes), mode, log2, shape);
dataset.key_bytes = day_bytes;
(p.to_string(), Epoch { program, dataset })
})
.collect()
});
&all.iter().find(|(n, _)| n == pack).unwrap().1
}
fn day_label(pack: &str) -> String {
json(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string()
}
fn check_program_json(pack: &str) {
let j = json(pack, "program.json");
let p = &epoch(pack).program;
assert_eq!(j["format"].as_str().unwrap(), "igneum-program-pack-3");
assert_eq!(j["generator"].as_u64().unwrap() as u32, p.generator, "{pack}: generator version");
assert_eq!(p.generator, if PACKS_V3.contains(&pack) { GENERATOR_VERSION_V3 } else { GENERATOR_VERSION });
assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt, "{pack}: attempt");
assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id");
let sw: Vec<u32> = j["seed_words"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(p.seed.to_vec(), sw, "{pack}: seed words");
assert_eq!(p.loads_per_hash() as u64, j["loads_per_hash"].as_u64().unwrap(), "{pack}: loads per hash");
assert_eq!(p.loads_per_hash(), 8 * LOAD_SLOTS);
assert!(accept::check(p).is_ok(), "{pack}: the pack's program passes the acceptance rule");
let instrs = j["instructions"].as_array().unwrap();
assert_eq!(instrs.len(), p.instrs.len(), "{pack}: instruction count");
for (k, (ins, ji)) in p.instrs.iter().zip(instrs).enumerate() {
assert_eq!(ji["i"].as_u64().unwrap() as usize, k);
assert_eq!(Op::from_name(ji["op"].as_str().unwrap()).unwrap(), ins.op, "{pack} #{k} op");
assert_eq!(ji["dst"].as_u64().unwrap(), ins.dst as u64, "{pack} #{k} dst");
assert_eq!(ji["src"].as_u64().unwrap(), ins.src as u64, "{pack} #{k} src");
assert_eq!(ji["src2"].as_u64().unwrap(), ins.src2 as u64, "{pack} #{k} src2");
assert_eq!(hex32(&ji["imm"]), ins.imm, "{pack} #{k} imm");
assert_eq!(hex32(&ji["imm2"]), ins.imm2, "{pack} #{k} imm2");
assert_eq!(ji["rot"].as_u64().unwrap(), ins.rot as u64, "{pack} #{k} rot");
assert_eq!(ji["bit"].as_u64().unwrap(), ins.bit as u64, "{pack} #{k} bit");
assert_eq!(ji["mask"].as_u64().unwrap(), ins.mask as u64, "{pack} #{k} mask");
}
let mix = j["op_mix"].as_object().unwrap();
for (name, count) in p.histogram() {
assert_eq!(mix[name].as_u64().unwrap() as usize, count, "{pack}: op_mix {name}");
}
assert_eq!(mix.len(), p.histogram().len());
}
#[test]
fn program_json_matches_all_packs() {
for pack in PACKS {
check_program_json(pack);
}
}
/// The genesis program of generator v2 (spec 01 section 1.4.3 vector).
#[test]
fn genesis_program_shape() {
let p = &epoch("igneum-genesis-mh").program;
assert_eq!(p.attempt, 0);
assert_eq!(p.op_mix(), "load=16 add=8 shfl=8 xor=6 mad=5 mul=5 mulhi=5 sub=4 rotl=3 rotr=3 or=1");
assert_eq!(p.program_id(), 0xbcc1248b10cc90f2);
assert_eq!(&epoch("igneum-genesis").program, p, "closed-form and memory-hard packs share the program");
}
#[test]
fn mixer_params_match_pack() {
for pack in ["igneum-genesis-mh", "igneum-devnet-v4-epoch0", "mx8-genesis", "mx8-devnet-epoch0", "mx4-genesis", "mx4-devnet-epoch0"] {
let j = json(pack, "program.json");
let mp = &epoch(pack).dataset.memhard().unwrap().params;
let key: Vec<u32> = j["dataset"]["key"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(mp.key.to_vec(), key);
let rot: Vec<u32> =
j["dataset"]["mixer"]["rot"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u32).collect();
assert_eq!(mp.rot.to_vec(), rot);
let mul: Vec<u32> = j["dataset"]["mixer"]["mul"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(mp.mul.to_vec(), mul);
let rc: Vec<u32> = j["dataset"]["mixer"]["rc"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(mp.rc.to_vec(), rc);
assert_eq!(hex32(&j["dataset"]["d0"]), mp.key[0]);
assert_eq!(hex32(&j["dataset"]["d1"]), mp.key[1]);
}
}
#[test]
fn cache_matches_vectors() {
let v = json("igneum-genesis-mh", "vectors.json");
let m = epoch("igneum-genesis-mh").dataset.memhard().unwrap();
let w = m.cache.words();
assert_eq!(w.len(), CACHE_WORDS);
let head: Vec<u32> = v["cache_head"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(&w[..16], &head[..]);
let last: Vec<u32> = v["cache_last_line"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(&w[CACHE_WORDS - 16..], &last[..]);
assert_eq!(m.cache.fnv1a64(), 0x48c4f5bf24166b2e, "cache FNV-1a 64 of day 2026-10-03 (MEMHARD.md), unchanged by v2");
assert_eq!(m.cache.fnv1a64(), hex64(&v["cache_fnv1a64"]));
let v = json("igneum-devnet-v4-epoch0", "vectors.json");
let m = epoch("igneum-devnet-v4-epoch0").dataset.memhard().unwrap();
assert_eq!(m.cache.fnv1a64(), hex64(&v["cache_fnv1a64"]));
assert_eq!(m.cache.fnv1a64(), 0x448274a57f508cbc, "cache FNV-1a 64 of day bytes igneum-day/20730");
}
fn check_dataset_words(pack: &str) {
let v = json(pack, "vectors.json");
let e = epoch(pack);
let ds = &e.dataset;
// a pack's self-test words are read under its program's layout (the era layout; linear for every v2 pack)
let head: Vec<u32> = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect();
for (i, h) in head.iter().enumerate() {
assert_eq!(e.dataset_word(i as u32), *h, "{pack}: dataset[{i}]");
}
let last_index = v["dataset_last_index"].as_u64().unwrap() as u32;
assert_eq!(last_index, ds.mask);
assert_eq!(e.dataset_word(last_index), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]");
let samples = v["dataset_samples"].as_array().unwrap();
assert_eq!(samples.len(), 64);
for s in samples {
let idx = s["index"].as_u64().unwrap() as u32;
assert_eq!(e.dataset_word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]");
}
}
#[test]
fn dataset_words_match_all_packs() {
for pack in PACKS {
check_dataset_words(pack);
}
}
/// Returns the number of hashes compared (3 warps x 32 lanes = 96).
fn check_vectors(pack: &str) -> usize {
let v = json(pack, "vectors.json");
let e = epoch(pack);
assert_eq!(v["dataset_mode"].as_str().unwrap(), e.dataset.mode().name());
assert_eq!(v["dataset_log2_words"].as_u64().unwrap() as u32, e.dataset.log2_words);
let warps = v["warps"].as_array().unwrap();
assert_eq!(warps.len(), 3);
let mut n = 0;
for w in warps {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let expected: Vec<u64> = w["expected"].as_array().unwrap().iter().map(hex64).collect();
let got = e.hash_warp(base);
for lane in 0..32 {
assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}");
n += 1;
}
// The single-nonce API agrees with the warp.
assert_eq!(e.hash(base + 7), expected[7]);
assert!(e.verify_block(base + 7, expected[7]));
assert!(!e.verify_block(base + 7, expected[7] - 1));
}
n
}
#[test]
fn vectors_96_per_pack() {
for pack in PACKS {
assert_eq!(check_vectors(pack), 96, "{pack}");
}
}
/// The spec 01 section 1.17 vectors (igneum-genesis-mh, generator v2).
#[test]
fn spec_vectors_genesis_mh() {
let e = epoch("igneum-genesis-mh");
let w = e.hash_warp(0);
assert_eq!(w[0], 0x42246ba99fc58e4f);
assert_eq!(w[31], 0xb08446b1f2de7793);
assert_eq!(e.hash_warp(4096)[0], 0x3d3903e310ca038f);
assert_eq!(e.hash_warp(1_000_000)[0], 0xf218c1bd58e6dfe0);
}
fn assert_same_text(pack: &str, file: &str, got: &str) {
let want = read(pack, file);
if got != want {
let (gl, wl): (Vec<&str>, Vec<&str>) = (got.lines().collect(), want.lines().collect());
for i in 0..gl.len().max(wl.len()) {
let g = gl.get(i).copied().unwrap_or("<eof>");
let w = wl.get(i).copied().unwrap_or("<eof>");
if g != w {
panic!("{pack}/{file} differs at line {}:\n pack: {w}\n rust: {g}", i + 1);
}
}
panic!("{pack}/{file} differs only in trailing bytes (len {} vs {})", got.len(), want.len());
}
}
fn check_sources(pack: &str) {
let e = epoch(pack);
let p = &e.program;
let day = day_label(pack);
let mp = e.dataset.memhard().map(|m| &m.params);
assert_same_text(pack, "kernel.cu", &cuda_kernel(p, mp));
assert_same_text(pack, "kernel_bound.cu", &cuda_kernel_bound(p, mp));
assert_same_text(pack, "program.metal", &metal_program(p, e.dataset.log2_words, LoadSource::Stored));
assert_same_text(pack, "program_bound.metal", &metal_program_bound(p, e.dataset.log2_words));
assert_same_text(pack, "kernel.cl", &opencl_kernel(p, mp));
assert_same_text(pack, "kernel_bound.cl", &opencl_kernel_bound(p, mp));
assert_same_text(pack, "program.h", &program_header(p, &day, &e.dataset));
if let Some(mp) = mp {
assert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp));
assert_same_text(pack, "memhard.metal", &igneum_pow::emit::metal_memhard_layout(mp, p.class.layout()));
}
let got = program_json(p, &day, &e.dataset);
assert_same_text(pack, "program.json", &got);
let _: Value = serde_json::from_str(&got).expect("program.json is valid JSON");
// Every load in every emitted hash kernel has the masked form, and there are exactly 16 of them (a class with
// the era layout inside has the era form instead: `((rotl_imm(rN * M, R) & WM) | OFF) & mask`, checked by
// era_emitted_sources_match_and_loads_have_the_era_form over the era packs, and here by the same count).
let era_load = if p.class.era.is_some() { "((rotl_imm(r" } else { "" };
for (file, load, masked) in [
("kernel.cu", "ds[r", " & mask]"),
("kernel_bound.cu", "ds[r", " & mask]"),
("program.metal", "dataset[r", " & MASK]"),
("program_bound.metal", "dataset[r", " & MASK]"),
] {
let text = read(pack, file);
let load = if era_load.is_empty() { load } else { era_load };
assert_eq!(text.matches(load).count(), LOAD_SLOTS, "{pack}/{file}: 16 loads");
assert_eq!(text.matches(masked).count(), LOAD_SLOTS, "{pack}/{file}: 16 masked loads");
}
}
#[test]
fn emitted_sources_match_all_packs() {
for pack in PACKS {
check_sources(pack);
}
}
/// The whole pack as `export` writes it: vectors.json and vectors.h match, and the file list is the full set.
/// The class v3 packs (Counter ASIC 2.0, `docs/plans/mixer-x4.md`): generator 3 on V3_CLASS = mx8; the program of
/// each is the v2 program of the same seed instruction for instruction (v2 loads take no width roll); the cache is
/// the v2 cache (day 0 of the growth rule: 2^26 words, the same FNV-1a 64); the dataset words differ from v2's;
/// program.json, program.h and the emitted memhard core say so; the id carries generator 3.
#[test]
fn v3_packs_are_the_v2_seeds_under_mixer_x8() {
assert_eq!(V3_CLASS.name(), "mx8");
assert_eq!(V3_CLASS.mixer_mult, 8);
assert!(V3_CLASS.growth);
assert_eq!(LoadClass { mixer_mult: 4, ..V3_CLASS }, LoadClass::MX4, "the x4 candidate differs from v3 in the multiplier alone");
for (v3, v2) in [("mx8-genesis", "igneum-genesis-mh"), ("mx8-devnet-epoch0", "igneum-devnet-v4-epoch0")] {
let e3 = epoch(v3);
let e2 = epoch(v2);
let j = json(v3, "program.json");
assert_eq!(j["program_class"].as_str().unwrap(), "v3");
assert!(j["load_class"].as_str().unwrap().starts_with("mx8"), "{v3}: mx8, or mx8 with the era inside");
assert_eq!(j["mixer_mult"].as_u64().unwrap(), 8);
assert_eq!(j["cache_growth"].as_bool().unwrap(), true);
assert_eq!(j["dataset"]["mixer_mult"].as_u64().unwrap(), 8);
assert_eq!(j["dataset"]["cache"]["log2_words"].as_u64().unwrap(), 26);
assert_eq!(e3.program.generator, GENERATOR_VERSION_V3);
if e3.program.era_bytes.is_some() {
// the era layout composed into class v3 (docs/plans/era-layout.md, 5 October 2026): a chain pack carries an
// era, so its class is V3_CLASS with the era drawn inside and its stream takes two window draws per
// instruction; the v2 program carries over only in the seed, the attempt and the day
assert_eq!(LoadClass { era: None, ..e3.program.class }, V3_CLASS, "{v3}: the composed class");
assert!(e3.program.class.era.is_some());
assert_ne!(e3.program.instrs, e2.program.instrs, "{v3}: the era windows change the stream");
} else {
assert_eq!(e3.program.class, V3_CLASS);
assert_eq!(e3.program.instrs, e2.program.instrs, "{v3}: the v2 program under the v3 construction");
}
assert_eq!(e3.program.seed, e2.program.seed);
assert_eq!(e3.program.attempt, e2.program.attempt);
assert_ne!(e3.program.program_id(), e2.program.program_id());
assert_eq!(e3.program.program_id(), igneum_pow::generator::program_id(GENERATOR_VERSION_V3, &e3.program.seed, e3.program.attempt));
let m3 = e3.dataset.memhard().unwrap();
let m2 = e2.dataset.memhard().unwrap();
assert_eq!(m3.shape(), Shape { mixer_mult: 8, cache_log2_words: 26, derive_len: 0, state: false });
assert_eq!(m3.cache.fnv1a64(), m2.cache.fnv1a64(), "{v3}: the same cache as v2 on day 0");
assert_eq!(m3.params.rot, m2.params.rot);
assert_eq!(e3.dataset.log2_words, 28);
assert_ne!(e3.dataset.word(0), e2.dataset.word(0), "{v3}: the dataset words differ");
assert_ne!(e3.hash_warp(0), e2.hash_warp(0));
let h = read(v3, "program.h");
assert!(h.contains("#define IGNEUM_GENERATOR 3\n"));
assert!(h.contains("#define IGNEUM_PROGRAM_CLASS \"v3\"\n"));
assert!(h.contains("#define IGNEUM_MIXER_MULT 8"));
assert!(h.contains("#define IGNEUM_CACHE_GROWTH 1"));
assert!(h.contains("#define IGNEUM_CACHE_LOG2_WORDS 26\n"));
assert!(h.contains("#define IGNEUM_LOAD_CLASS \"mx8"), "mx8, or mx8 with the era inside");
for file in ["memhard.h", "memhard.metal", "kernel.cl"] {
let text = read(v3, file);
assert_eq!(text.matches("j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 8u + j + 1u))").count(), 1, "{v3}/{file}");
assert_eq!(text.matches("j < 8u; ++j) mh_mixer(s, 0x9E3779B9u * (64u + j + 1u))").count(), 1, "{v3}/{file}");
}
for file in ["memhard.h", "memhard.metal", "kernel.cl"] {
let text = read(v2, file);
assert_eq!(text.matches("j < 8u").count(), 0, "{v2}/{file}: the v2 text has no multiplier loop");
}
}
// the devnet v3 pack records era 0's stand-in, the devnet genesis hash
let j = json("mx8-devnet-epoch0", "program.json");
assert_eq!(unhex(&j["era_seed_bytes"]), unhex(&j["seed_bytes"]));
assert!(read("mx8-devnet-epoch0", "program.h").contains("#define IGNEUM_ERA_SEED_HEX \"edc4fa844da9dc98"));
assert!(json("mx8-genesis", "program.json").get("era_seed_bytes").is_none());
// the x4 candidate's packs: generator 2, the class in the id, the v2 program of the seed, mixer x4
for (x4, v2) in [("mx4-genesis", "igneum-genesis-mh"), ("mx4-devnet-epoch0", "igneum-devnet-v4-epoch0")] {
let e4 = epoch(x4);
let j = json(x4, "program.json");
assert_eq!(e4.program.generator, GENERATOR_VERSION);
assert_eq!(j["load_class"].as_str().unwrap(), "mx4");
assert_eq!(e4.program.class, LoadClass::MX4);
assert_eq!(e4.program.instrs, epoch(v2).program.instrs);
assert_eq!(e4.dataset.memhard().unwrap().shape(), Shape { mixer_mult: 4, cache_log2_words: 26, derive_len: 0, state: false });
assert!(read(x4, "memhard.h").contains("j < 4u; ++j) mh_mixer(s, 0x9E3779B9u * (r * 4u + j + 1u))"));
}
}
fn check_export(pack: &str) {
let e = epoch(pack);
let v = json(pack, "vectors.json");
let source = v["source"].as_str().unwrap();
let out = export_pack(e, &day_label(pack), source);
let file = |name: &str| -> &str { &out.files.iter().find(|(n, _)| n == name).unwrap().1 };
assert_same_text(pack, "vectors.json", file("vectors.json"));
assert_same_text(pack, "vectors.h", file("vectors.h"));
let mut expected = vec![
"program.json",
"vectors.json",
"kernel.cu",
"kernel.cl",
"program.h",
"vectors.h",
"program.metal",
"program_bound.metal",
"kernel_bound.cu",
"kernel_bound.cl",
];
if e.dataset.mode() == DatasetMode::MemoryHard {
expected.extend(["memhard.h", "memhard.metal"]);
}
assert_eq!(out.files.iter().map(|(n, _)| n.as_str()).collect::<Vec<_>>(), expected);
let mut on_disk: Vec<String> = std::fs::read_dir(pack_dir(pack))
.unwrap()
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
.filter(|n| !n.starts_with('.'))
.collect();
on_disk.sort();
let mut want: Vec<String> = expected.iter().map(|s| s.to_string()).collect();
want.sort();
assert_eq!(on_disk, want, "{pack}: no stale file in the pack directory");
}
#[test]
fn export_pack_matches_all_packs() {
for pack in PACKS {
check_export(pack);
}
}
#[test]
fn item_is_independent_of_dataset_size() {
// MEMHARD.md 1.7: a smaller dataset is a prefix of items, so dataset[w] is the same at every size.
let big = &epoch("igneum-genesis-mh").dataset;
let m = big.memhard().unwrap();
for w in [0u32, 1, 15, 16, 17, 0x00ff_ffff, 0x03ff_ffff] {
assert_eq!(big.word(w), m.word(w));
}
}
/// The devnet pack is the chain's own derivation: epoch seed = the devnet genesis hash, day bytes = `bind::day_bytes(20730)`.
#[test]
fn devnet_pack_is_the_chain_derivation() {
let j = json("igneum-devnet-v4-epoch0", "program.json");
let genesis = igneum_pow::bind::unhex("edc4fa844da9dc98d37e965176f6558a31560e40502ab3ae5491b21aaaabfb07").unwrap();
assert_eq!(unhex(&j["seed_bytes"]), genesis);
assert_eq!(unhex(&j["dataset"]["day_bytes"]), igneum_pow::bind::day_bytes(20_730).to_vec());
let e = Epoch::from_seed_bytes(&genesis, &igneum_pow::bind::day_bytes(20_730), "devnet");
assert_eq!(e.program.instrs, epoch("igneum-devnet-v4-epoch0").program.instrs);
assert_eq!(e.hash_warp(0), epoch("igneum-devnet-v4-epoch0").hash_warp(0));
}
// ---------------------------------------------------------------------------------------------------------
// Era layout packs (5 October 2026, docs/plans/era-layout.md): proto-cuda/packs-ca2-era/era-<n>, n in 0..5, the
// devnet epoch seed and day bytes under the era class of test era seed igneum-era-test/<n>, the width pinned at
// 4 bytes (allowed_widths in program.json). Checked like the pinned packs, plus the one load form of 1.3 by text search.
// ---------------------------------------------------------------------------------------------------------
use igneum_pow::generator::{generate_era, EraParams, V3_ALLOWED};
const ERA_PACKS: [&str; 6] = ["era-0", "era-1", "era-2", "era-3", "era-4", "era-5"];
fn era_packs_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-era")
}
fn era_read(pack: &str, file: &str) -> String {
let p = era_packs_dir().join(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
fn era_json(pack: &str, file: &str) -> Value {
serde_json::from_str(&era_read(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"))
}
/// The era class of a pack: the era bytes from program.json (`era_seed_bytes`, the chain's `E_n`; the test seed
/// `igneum-era-test/<n>` of the pack's number gives the same bytes), the allowed set from `era.allowed_widths`
/// (class v3's `V3_ALLOWED`); the pack's recorded stream words and draw must be the class's.
fn era_class(pack: &str) -> LoadClass {
let j = era_json(pack, "program.json");
let n: u64 = pack.trim_start_matches("era-").parse().unwrap();
let eb = unhex(&j["era_seed_bytes"]);
assert_eq!(eb, EraParams::test_era_bytes(&format!("igneum-era-test/{n}")).to_vec(), "{pack}: the era bytes of test seed {n}");
let allowed: Vec<u8> = j["era"]["allowed_widths"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u8).collect();
assert_eq!(allowed, V3_ALLOWED.to_vec(), "{pack}: class v3's width set");
// the measurement packs of 5 October 2026: the era layout over version 2's construction (mixer x1, the genesis
// cache), generator 3 and the era bytes recorded; the chain's class v3 composes the same draw over LoadClass::MX4
// (generator tests era_programs_are_accepted and program_classes), and the integration re-exports these packs
let c = LoadClass::era(LoadClass::V2, &eb, &allowed);
let e = c.era.unwrap();
let words: Vec<u32> = j["era"]["seed_words"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(e.words.to_vec(), words, "{pack}: era seed words");
assert_eq!(e.width_words as u64, j["era"]["width_words"].as_u64().unwrap(), "{pack}: width");
assert_eq!(e.stride_mul, hex32(&j["era"]["stride_mul"]), "{pack}: stride mul");
assert_eq!(e.stride_rot as u64, j["era"]["stride_rot"].as_u64().unwrap(), "{pack}: stride rot");
let pos: Vec<u8> = j["era"]["interleave"].as_array().unwrap().iter().map(|v| v.as_u64().unwrap() as u8).collect();
assert_eq!(e.pos.to_vec(), pos, "{pack}: interleave");
c
}
fn era_epoch(pack: &str) -> &'static Epoch {
static E: OnceLock<Vec<(String, Epoch)>> = OnceLock::new();
let all = E.get_or_init(|| {
ERA_PACKS
.iter()
.map(|p| {
let j = era_json(p, "program.json");
let seed = j["seed"].as_str().unwrap();
let seed_bytes = unhex(&j["seed_bytes"]);
let day_bytes = unhex(&j["dataset"]["day_bytes"]);
assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard");
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
let class = era_class(p);
assert_eq!(log2, igneum_pow::verify::DEFAULT_DATASET_LOG2);
let eb = unhex(&j["era_seed_bytes"]);
let program = generate_era(seed, &seed_bytes, LoadClass::V2, &eb, &V3_ALLOWED);
assert_eq!(program.class, class, "{p}: the pack's era class");
assert_eq!(program.generator, GENERATOR_VERSION_V3);
assert_eq!(program.era_bytes.as_deref(), Some(&eb[..]));
let mut dataset = DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2);
dataset.key_bytes = day_bytes;
(p.to_string(), Epoch { program, dataset })
})
.collect()
});
&all.iter().find(|(n, _)| n == pack).unwrap().1
}
/// program.json of an era pack: generator, attempt, id, class, every instruction with width, win and off, and the
/// program passes the acceptance rule.
#[test]
fn era_program_json_matches() {
for pack in ERA_PACKS {
let j = era_json(pack, "program.json");
let p = &era_epoch(pack).program;
assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION_V3, "{pack}: a class v3 pack");
assert_eq!(j["program_class"].as_str().unwrap(), "v3");
assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt, "{pack}: attempt");
assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id");
assert_eq!(j["load_class"].as_str().unwrap(), p.class.name(), "{pack}: class");
assert_eq!(j["bytes_per_hash"].as_u64().unwrap() as usize, p.bytes_per_hash());
assert_eq!(p.loads_per_hash(), 8 * LOAD_SLOTS);
assert!(accept::check(p).is_ok(), "{pack}: acceptance");
let instrs = j["instructions"].as_array().unwrap();
assert_eq!(instrs.len(), p.instrs.len());
for (k, (ins, ji)) in p.instrs.iter().zip(instrs).enumerate() {
assert_eq!(Op::from_name(ji["op"].as_str().unwrap()).unwrap(), ins.op, "{pack} #{k} op");
assert_eq!(ji["dst"].as_u64().unwrap(), ins.dst as u64);
assert_eq!(ji["src"].as_u64().unwrap(), ins.src as u64);
assert_eq!(hex32(&ji["imm"]), ins.imm);
assert_eq!(ji["width"].as_u64().unwrap(), ins.width as u64, "{pack} #{k} width");
assert_eq!(ji["win"].as_u64().unwrap(), ins.win as u64, "{pack} #{k} win");
assert_eq!(ji["off"].as_u64().unwrap(), ins.off as u64, "{pack} #{k} off");
if ins.op == Op::Load {
assert_eq!(ins.width, p.class.era.unwrap().width_words);
assert!(ins.win <= 2 && (ins.off as u32) < (1u32 << ins.win));
}
}
}
}
/// The six era packs are the same program seed under six draws: the instruction lists agree, the widths and layouts
/// follow the draw, and the dataset words differ from the linear layout exactly when the interleave is not linear.
#[test]
fn era_packs_share_the_program_and_differ_in_layout() {
let linear = &epoch("igneum-devnet-v4-epoch0").dataset;
for pack in ERA_PACKS {
let e = era_epoch(pack);
assert_eq!(e.program.seed_bytes, epoch("igneum-devnet-v4-epoch0").program.seed_bytes, "{pack}: the devnet seed");
let strip = |p: &igneum_pow::generator::Program| {
p.instrs.iter().map(|i| (i.op, i.dst, i.src, i.src2, i.imm, i.imm2, i.rot, i.bit, i.mask, i.win, i.off)).collect::<Vec<_>>()
};
assert_eq!(strip(&e.program), strip(&era_epoch("era-0").program), "{pack}: same stream as era-0");
let l = e.program.class.layout();
let same_at_1 = (0..64u32).all(|w| e.dataset_word(w * 977 + 1) == linear.word(w * 977 + 1));
assert_eq!(same_at_1, l.is_linear(), "{pack}: layout {:?}", l.pos);
assert_eq!(e.dataset_word(0), linear.word(0), "{pack}: word 0 is item 0 word 0 in every layout");
// the chain's shared day cache serves every era: the day's dataset source is the pinned pack's, bit for bit
assert_eq!(e.dataset.key, linear.key);
assert_eq!(e.dataset.memhard().unwrap().cache.fnv1a64(), linear.memhard().unwrap().cache.fnv1a64());
}
}
#[test]
fn era_dataset_words_and_vectors_match() {
for pack in ERA_PACKS {
let v = era_json(pack, "vectors.json");
let e = era_epoch(pack);
let ds = &e.dataset;
let head: Vec<u32> = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect();
for (i, h) in head.iter().enumerate() {
assert_eq!(e.dataset_word(i as u32), *h, "{pack}: dataset[{i}]");
}
assert_eq!(e.dataset_word(ds.mask), hex32(&v["dataset_last"]), "{pack}: dataset[MASK]");
for s in v["dataset_samples"].as_array().unwrap() {
let idx = s["index"].as_u64().unwrap() as u32;
assert_eq!(e.dataset_word(idx), hex32(&s["value"]), "{pack}: dataset[{idx}]");
}
assert_eq!(ds.memhard().unwrap().cache.fnv1a64(), hex64(&v["cache_fnv1a64"]));
let mut n = 0;
for w in v["warps"].as_array().unwrap() {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let expected: Vec<u64> = w["expected"].as_array().unwrap().iter().map(hex64).collect();
let got = e.hash_warp(base);
for lane in 0..32 {
assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}");
n += 1;
}
assert_eq!(e.hash(base + 7), expected[7]);
}
assert_eq!(n, 96, "{pack}");
}
}
/// Every emitted file of every era pack matches the emitters byte for byte, the export reproduces vectors.json and
/// vectors.h, and every dataset load in every hash kernel has the one era form (no plain `ds[rN & mask]` remains).
#[test]
fn era_emitted_sources_match_and_loads_have_the_era_form() {
for pack in ERA_PACKS {
let e = era_epoch(pack);
let day = era_json(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string();
let source = era_json(pack, "vectors.json")["source"].as_str().unwrap().to_string();
let out = export_pack(e, &day, &source);
for (name, text) in &out.files {
let want = era_read(pack, name);
assert!(text == &want, "{pack}/{name} differs from the emitter");
}
let mut on_disk: Vec<String> = std::fs::read_dir(era_packs_dir().join(pack))
.unwrap()
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
.filter(|n| !n.starts_with('.') && n != "seeds.txt")
.collect();
on_disk.sort();
let mut want: Vec<String> = out.files.iter().map(|(n, _)| n.clone()).collect();
want.sort();
assert_eq!(on_disk, want, "{pack}: the pack holds the export's files and seeds.txt only");
let era = e.program.class.era.unwrap();
let mul = format!("0x{:08x}u", era.stride_mul);
for (file, mask) in [
("kernel.cu", "mask"),
("kernel_bound.cu", "mask"),
("kernel.cl", "mask"),
("kernel_bound.cl", "mask"),
("program.metal", "MASK"),
("program_bound.metal", "MASK"),
] {
let text = era_read(pack, file);
let kernels = if file.starts_with("kernel_bound") || file == "kernel.cu" || file == "kernel.cl" || file.starts_with("program") { 1 } else { 1 };
// kernel_bound.cl carries igneum_hash and igneum_hash_bound: two kernels
let kernels = if file == "kernel_bound.cl" { 2 } else { kernels };
let era_form: usize = text
.lines()
.filter(|l| l.contains("rotl_imm(r") && l.contains(&format!(" * {mul}, {}u) & ", era.stride_rot)) && l.contains(&format!(") & {mask}")))
.filter(|l| l.contains("ds[") || l.contains("dataset[") || l.contains("b_ = "))
.count();
assert_eq!(era_form, LOAD_SLOTS * kernels, "{pack}/{file}: {} loads of the era form", LOAD_SLOTS * kernels);
let plain = text.lines().filter(|l| l.contains("ds[r") || l.contains("dataset[r")).count();
assert_eq!(plain, 0, "{pack}/{file}: a load without the era form");
}
// the layout helpers appear exactly when the layout is not linear
let mh = era_read(pack, "memhard.h");
assert_eq!(mh.contains("mh_addr("), !era.layout().is_linear(), "{pack}: memhard.h layout helpers");
}
}
/// An era pack's dataset is a prefix at every size of at least 2^16 words: the 2^20-word source gives the pack's
/// words below 2^20.
#[test]
fn era_dataset_is_a_prefix_at_smaller_sizes() {
for pack in ["era-1", "era-3"] {
let e = era_epoch(pack);
let small = DatasetSource::from_key(e.dataset.key, DatasetMode::MemoryHard, 20);
let l = e.program.class.layout();
for w in [0u32, 1, 2, 3, 16, 255, 4096, 65_535, 65_536, 0x000f_ffff] {
assert_eq!(small.word_at(l, w), e.dataset_word(w), "{pack}: w {w}");
}
}
}
// ---------------------------------------------------------------------------------------------------------
// Hot-table experiment (5 October 2026, docs/plans/hot-table.md): the five packs under proto-cuda/packs-ca2-hot/ are
// pinned the same way (program, vectors, every emitted file byte for byte), plus the hot table's fingerprint and the
// one-form load check: exactly 16 - k masked dataset loads and exactly k hot loads in every hash kernel.
// ---------------------------------------------------------------------------------------------------------
const HOT_PACKS: [&str; 8] = ["hot32k4", "hot64k4", "hot96k4", "hot64k2", "hot64k8", "hot32k4a", "hot64k4a", "hot96k4a"];
fn hot_packs_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca2-hot")
}
fn hread(pack: &str, file: &str) -> String {
let p = hot_packs_dir().join(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
fn hjson(pack: &str, file: &str) -> Value {
serde_json::from_str(&hread(pack, file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"))
}
/// The epoch of a hot pack from program.json alone: the class from `load_class`, the program from the seed bytes,
/// the dataset from the day bytes, the hot table from the seed bytes (what `Epoch::from_seed_bytes_class` does).
fn hepoch(pack: &str) -> &'static Epoch {
static E: OnceLock<Vec<(String, Epoch)>> = OnceLock::new();
let all = E.get_or_init(|| {
HOT_PACKS
.iter()
.map(|p| {
let j = hjson(p, "program.json");
let seed = j["seed"].as_str().unwrap();
let seed_bytes = unhex(&j["seed_bytes"]);
let day_bytes = unhex(&j["dataset"]["day_bytes"]);
assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard");
let class = LoadClass::parse(j["load_class"].as_str().unwrap()).unwrap();
assert_eq!(class.name(), *p, "the pack directory is the class name");
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
let program = generate_from_seed_bytes_class(seed, &seed_bytes, class);
let mut dataset =
DatasetSource::from_key(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2);
dataset.key_bytes = day_bytes;
dataset.attach_hot_for(&program);
(p.to_string(), Epoch { program, dataset })
})
.collect()
});
&all.iter().find(|(n, _)| n == pack).unwrap().1
}
fn hassert_same_text(pack: &str, file: &str, got: &str) {
let want = hread(pack, file);
if got != want {
let (gl, wl): (Vec<&str>, Vec<&str>) = (got.lines().collect(), want.lines().collect());
for i in 0..gl.len().max(wl.len()) {
let g = gl.get(i).copied().unwrap_or("<eof>");
let w = wl.get(i).copied().unwrap_or("<eof>");
if g != w {
panic!("{pack}/{file} differs at line {}:\n pack: {w}\n rust: {g}", i + 1);
}
}
panic!("{pack}/{file} differs only in trailing bytes (len {} vs {})", got.len(), want.len());
}
}
#[test]
fn hot_packs_program_and_vectors() {
for pack in HOT_PACKS {
let e = hepoch(pack);
let p = &e.program;
let j = hjson(pack, "program.json");
let h = p.class.hot.unwrap();
assert_eq!(j["generator"].as_u64().unwrap() as u32, GENERATOR_VERSION);
assert_eq!(j["attempt"].as_u64().unwrap() as u32, p.attempt);
assert_eq!(hex64(&j["program_id"]), p.program_id(), "{pack}: program id");
let dataset_slots = if h.added { 16 } else { 16 - h.k as usize };
assert_eq!(j["loads_per_hash"].as_u64().unwrap() as usize, (dataset_slots + h.k as usize) * 8);
assert_eq!(j["hot_table"]["mb"].as_u64().unwrap(), h.mb as u64);
assert_eq!(j["hot_table"]["slots"].as_u64().unwrap(), h.k as u64);
assert_eq!(j["hot_table"]["dataset_slots"].as_u64().unwrap() as usize, dataset_slots);
assert_eq!(j["hot_table"]["words"].as_u64().unwrap() as u32, p.hot_words());
assert_eq!(j["op_mix"]["hot"].as_u64().unwrap(), h.k as u64, "{pack}: k hot instructions");
assert_eq!(j["op_mix"]["load"].as_u64().unwrap() as usize, dataset_slots);
assert_eq!(p.items_per_warp(), dataset_slots * 8 * 32);
assert!(accept::check(p).is_ok(), "{pack}: passes the acceptance rule");
let v2 = &epoch("igneum-genesis-mh").program;
if !h.added {
// replaced form: the version 2 genesis program with k loads redirected (attempt 0 on both)
assert_eq!(p.attempt, v2.attempt);
for (a, b) in p.instrs.iter().zip(v2.instrs.iter()) {
if a.op == Op::Hot {
assert_eq!(b.op, Op::Load);
} else {
assert_eq!(a, b);
}
}
} else {
// added form: 16 + k load slots, so another slot draw and another program; 16 dataset loads stay
assert_ne!(p.instrs, v2.instrs);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count(), 16);
assert!(p.instrs.iter().all(|i| i.width == 1));
}
// the hot table: the pack's head, last line and fingerprint
let v = hjson(pack, "vectors.json");
let t = e.dataset.hot.as_ref().unwrap();
assert_eq!(t.n_words(), p.hot_words());
let head: Vec<u32> = v["hot_head"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(&t.words()[..16], &head[..]);
let last: Vec<u32> = v["hot_last_line"].as_array().unwrap().iter().map(hex32).collect();
assert_eq!(&t.words()[t.words().len() - 16..], &last[..]);
assert_eq!(t.fnv1a64(), hex64(&v["hot_fnv1a64"]), "{pack}: hot_fnv1a64");
assert_eq!(t.key, igneum_pow::memhard::hot_key(&p.seed_bytes));
// the cache is the day's, unchanged by the class
assert_eq!(e.dataset.memhard().unwrap().cache.fnv1a64(), 0x48c4f5bf24166b2e);
// 96 vectors
let warps = v["warps"].as_array().unwrap();
assert_eq!(warps.len(), 3);
for w in warps {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let expected: Vec<u64> = w["expected"].as_array().unwrap().iter().map(hex64).collect();
let got = e.hash_warp(base);
for lane in 0..32 {
assert_eq!(got[lane], expected[lane], "{pack}: base {base} lane {lane}");
}
assert_eq!(e.hash(base + 5), expected[5]);
}
// the dataset words are the day's
let head: Vec<u32> = v["dataset_head"].as_array().unwrap().iter().map(hex32).collect();
for (i, hd) in head.iter().enumerate() {
assert_eq!(e.dataset.word(i as u32), *hd);
}
}
// the same k at three sizes: identical programs, three fingerprints, three vector sets
let a = hepoch("hot32k4");
let b = hepoch("hot64k4");
let c = hepoch("hot96k4");
assert_eq!(a.program.instrs, b.program.instrs);
assert_eq!(b.program.instrs, c.program.instrs);
assert_ne!(a.hash_warp(0), b.hash_warp(0));
assert_ne!(b.hash_warp(0), c.hash_warp(0));
}
#[test]
fn hot_packs_emitted_sources_and_load_forms() {
for pack in HOT_PACKS {
let e = hepoch(pack);
let p = &e.program;
let k = p.class.hot.unwrap().k as usize;
let dataset_loads = p.class.dataset_slots();
let day = hjson(pack, "program.json")["dataset"]["day"].as_str().unwrap().to_string();
let mp = &e.dataset.memhard().unwrap().params;
hassert_same_text(pack, "kernel.cu", &cuda_kernel(p, Some(mp)));
hassert_same_text(pack, "kernel_bound.cu", &cuda_kernel_bound(p, Some(mp)));
hassert_same_text(pack, "program.metal", &metal_program(p, e.dataset.log2_words, LoadSource::Stored));
hassert_same_text(pack, "program_bound.metal", &metal_program_bound(p, e.dataset.log2_words));
hassert_same_text(pack, "kernel.cl", &opencl_kernel(p, Some(mp)));
hassert_same_text(pack, "kernel_bound.cl", &opencl_kernel_bound(p, Some(mp)));
hassert_same_text(pack, "program.h", &program_header(p, &day, &e.dataset));
hassert_same_text(pack, "memhard.h", &cuda_memhard_header(p, mp));
hassert_same_text(pack, "memhard.metal", &metal_memhard_for(p, mp));
assert_ne!(metal_memhard_for(p, mp), metal_memhard(mp), "{pack}: the hot fill kernel is in memhard.metal");
let got = program_json(p, &day, &e.dataset);
hassert_same_text(pack, "program.json", &got);
let _: Value = serde_json::from_str(&got).expect("program.json is valid JSON");
let v = hjson(pack, "vectors.json");
let out = export_pack(e, &day, v["source"].as_str().unwrap());
let file = |name: &str| -> &str { &out.files.iter().find(|(n, _)| n == name).unwrap().1 };
hassert_same_text(pack, "vectors.json", file("vectors.json"));
hassert_same_text(pack, "vectors.h", file("vectors.h"));
assert_eq!(out.files.len(), 12);
// One form per dialect, exactly 16 - k masked dataset loads and k hot loads in every hash kernel; the fill
// kernel is present once per source that builds the table.
for (file, load, masked, hot) in [
("kernel.cu", "ds[r", " & mask]", "hot[__umulhi(r"),
("kernel_bound.cu", "ds[r", " & mask]", "hot[__umulhi(r"),
("program.metal", "dataset[r", " & MASK]", "hot[mulhi(r"),
("program_bound.metal", "dataset[r", " & MASK]", "hot[mulhi(r"),
("kernel.cl", "ds[r", " & mask]", "hot[mul_hi(r"),
] {
let text = hread(pack, file);
assert_eq!(text.matches(load).count(), dataset_loads, "{pack}/{file}: {dataset_loads} dataset loads");
assert_eq!(text.matches(masked).count(), dataset_loads, "{pack}/{file}: masked loads");
assert_eq!(text.matches(hot).count(), k, "{pack}/{file}: {k} hot loads");
assert!(text.contains(&format!("#define HOT_WORDS 0x{:08x}u", p.hot_words())), "{pack}/{file}: HOT_WORDS literal");
}
// kernel_bound.cl carries both kernels
let text = hread(pack, "kernel_bound.cl");
assert_eq!(text.matches("hot[mul_hi(r").count(), 2 * k);
assert_eq!(text.matches("ds[r").count(), 2 * dataset_loads);
for file in ["kernel.cu", "kernel.cl", "kernel_bound.cl", "memhard.metal"] {
assert_eq!(hread(pack, file).matches("igneum_hot_fill(").count(), 1, "{pack}/{file}: one hot fill kernel");
}
assert_eq!(hread(pack, "memhard.h").matches("void ht_segment(").count(), 1);
let ph = hread(pack, "program.h");
assert!(ph.contains(&format!("#define IGNEUM_HOT_MB {}", p.class.hot.unwrap().mb)));
assert!(ph.contains(&format!("#define IGNEUM_HOT_SLOTS {k}")));
assert!(ph.contains("igneum_launch_hot_fill("));
}
}
// ---------------------------------------------------------------------------------------------------------------
// Class v5 (docs/design/class-v5-stored-state.md, 7 October 2026): the pinned pack under proto-cuda/packs-ca3-v5/
// ---------------------------------------------------------------------------------------------------------------
fn v5_packs_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-ca3-v5")
}
fn v5_read(pack: &str, file: &str) -> String {
std::fs::read_to_string(v5_packs_dir().join(pack).join(file)).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"))
}
fn v5_json(pack: &str, file: &str) -> Value {
serde_json::from_str(&v5_read(pack, file)).unwrap()
}
/// The epoch of a pinned class v4 or v5 pack of the string seed and day: the program through the seam, the dataset at
/// the day-0 size, and for a v5 pack the leaves of its `state.igsd1` (the devnet's state stream of 7 October 2026,
/// node 1's exec snapshot at chain block 159,357: 93 records, root 0x1c583d35...).
fn v5_epoch(pack: &str) -> Epoch {
let j = v5_json(pack, "program.json");
let seed = j["seed"].as_str().unwrap();
let class = ProgramClass::parse(j["program_class"].as_str().unwrap()).unwrap();
let program = generate_from_seed_bytes_program_class(seed, seed.as_bytes(), class, None);
let day = j["dataset"]["day"].as_str().unwrap();
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
let shape = Shape::for_class(&program.class);
let mut dataset = DatasetSource::new_shape(day, DatasetMode::MemoryHard, log2, shape);
if class == ProgramClass::V5 {
let stream = igneum_pow::StateStream::read_file(&v5_packs_dir().join(pack).join("state.igsd1")).unwrap();
dataset = dataset.with_leaves(std::sync::Arc::new(igneum_pow::StateLeaves::from_stream(&stream, log2)));
}
Epoch { program, dataset }
}
/// The class v5 pack is the class v4 program of the same seed over the state leaves: generator 5 and
/// `program_id(5, seed, attempt)`, the base program, the shadow block and the dataset's cache equal to the v4 pack's,
/// every vector and dataset word different, every emitted file byte for byte what the crate exports (leaves.bin
/// included, its FNV in program.h), and the hash kernel text of kernel.cu equal to the v4 pack's but for the build
/// kernel and the class lines. The known-failed case first: the v4 control pack's vectors under the v5 epoch agree on
/// no lane.
#[test]
fn v5_pack_is_the_v4_program_over_the_state_leaves() {
let e5 = v5_epoch("v5-genesis");
let e4 = v5_epoch("v4-genesis");
let v4 = v5_json("v4-genesis", "vectors.json");
// the known-failed case: the v4 pack's hashes are not the v5 epoch's on any lane
let out4: Vec<u64> = v4["warps"][0]["expected"].as_array().unwrap().iter().map(hex64).collect();
let got5 = e5.hash_warp(0);
assert_eq!(out4.iter().zip(got5.iter()).filter(|(a, b)| a == b).count(), 0, "0 of 32 lanes of the v4 pack agree with the v5 epoch");
assert_eq!(e4.hash_warp(0).to_vec(), out4, "the v4 control pack is the v4 epoch");
// the program: v4's draw, generator 5, the state in the class and the id
assert_eq!(e5.program.generator, igneum_pow::GENERATOR_VERSION_V5);
assert_eq!(e5.program.class, igneum_pow::V5_CLASS);
assert_eq!(e5.program.class.name(), "mx8+sh256x27+state");
assert_eq!(e5.program.instrs, e4.program.instrs);
assert_eq!(e5.program.shadow, e4.program.shadow);
assert_eq!((e5.program.seed, e5.program.attempt), (e4.program.seed, e4.program.attempt));
assert_eq!(e5.program.program_id(), igneum_pow::generator::program_id(igneum_pow::GENERATOR_VERSION_V5, &e5.program.seed, e5.program.attempt));
assert_ne!(e5.program.program_id(), e4.program.program_id());
// the dataset: the same cache, other items
let m5 = e5.dataset.memhard().unwrap();
let m4 = e4.dataset.memhard().unwrap();
assert_eq!(m5.cache.fnv1a64(), m4.cache.fnv1a64(), "one day cache");
assert!(m5.shape().state && !m4.shape().state);
let leaves = e5.dataset.leaves().unwrap();
assert_eq!((leaves.n(), leaves.records_total, leaves.sampled), (93, 93, false));
assert_ne!(e5.dataset_word(0), e4.dataset_word(0));
// every file as the crate exports it, leaves.bin included
for pack in ["v4-genesis", "v5-genesis"] {
let e = if pack == "v5-genesis" { &e5 } else { &e4 };
let v = v5_json(pack, "vectors.json");
let out = export_pack(e, v["day"].as_str().unwrap(), v["source"].as_str().unwrap());
for (name, text) in &out.files {
assert_eq!(v5_read(pack, name), *text, "{pack}/{name} differs from the export");
}
for (name, bytes) in &out.binaries {
assert_eq!(std::fs::read(v5_packs_dir().join(pack).join(name)).unwrap(), *bytes, "{pack}/{name}");
}
assert_eq!(out.binaries.len(), (pack == "v5-genesis") as usize);
}
let h5 = v5_read("v5-genesis", "program.h");
assert!(h5.contains("#define IGNEUM_PROGRAM_CLASS \"v5\"") && h5.contains("#define IGNEUM_STATE_LEAVES 93") && h5.contains(&format!("#define IGNEUM_STATE_LEAVES_FNV64 {}", igneum_pow::emit::hex64(leaves.fnv1a64()))));
assert!(h5.contains("igneum_launch_build(uint32_t* ds, const uint32_t* cache, const uint32_t* leaves, uint32_t nLeaves, uint32_t nItems)"));
// the hash kernel text is class v4's byte for byte; the build kernel and the class lines are what differ
let hash_text = |s: &str| {
let a = s.find("__global__ void igneum_hash(").unwrap();
let b = s.find("// Host-side launch wrappers").unwrap();
s[a..b].to_string()
};
let k5 = v5_read("v5-genesis", "kernel.cu");
let k4 = v5_read("v4-genesis", "kernel.cu");
assert_eq!(hash_text(&k5), hash_text(&k4), "the hash kernel is class v4's");
assert!(k5.contains("mh_item(cache, mh_leaf(leaves, nLeaves, t), t, s)") && !k4.contains("mh_leaf"));
let mh5 = v5_read("v5-genesis", "memhard.h");
assert!(mh5.contains("s[i] ^= leaf[i]"), "the leaf XOR before the first mixer");
}

View file

@ -0,0 +1,143 @@
//! The miner's CPU re-check against the worker's reference path, class v3 and class v4 (6 October 2026).
//!
//! The fleet's 10-member pool run (18:14Z to 18:24Z): 0 shares accepted in ten minutes, every GPU answer refused as
//! `WORKER MISMATCH`, because the pool fork's miner re-hashed each share with a fixed class v2 program while the 0.3.14
//! CUDA worker hashed the class v3 program of the epoch. The rule since: the re-check builds its program through the
//! chain seam (`Epoch::chain_program` and `Epoch::chain_dataset_day`, what the node's `IgneumEngine` calls) from the
//! template's `(epoch seed, day, program class, era seed)`, never from a fixed class.
//!
//! This test pins that seam against the worker's reference for both classes the packs carry: the program the pack's
//! kernels were emitted from (program.json, generator 3 or 4, with its id) and the pack's 96 vectors; then one known
//! nonce through the worker's path (`block_init_words` plus `interpret_warp_init` on the pack's program) and through
//! the re-check's path (the same on the seam's program and dataset) must agree, and the other class's program and
//! the fixed-v2 program of the same seed must disagree (the mismatch the fleet saw).
use igneum_pow::bind::{block_init_words, lane_nonce, unhex};
use igneum_pow::verify::DatasetSource;
use igneum_pow::{interpret_warp_init, DatasetMode, Epoch, ProgramClass, Shape};
use serde_json::Value;
use std::path::PathBuf;
fn pack_dir(rel: &str) -> PathBuf {
let p = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda").join(rel);
assert!(p.join("program.json").is_file(), "pack {} is missing; this test never skips", p.display());
p
}
fn json(dir: &std::path::Path, f: &str) -> Value {
serde_json::from_str(&std::fs::read_to_string(dir.join(f)).unwrap_or_else(|e| panic!("{}: {e}", dir.join(f).display()))).unwrap()
}
fn hex64(v: &Value) -> u64 {
u64::from_str_radix(v.as_str().unwrap().trim_start_matches("0x"), 16).unwrap()
}
struct PackRef {
class: ProgramClass,
seed_label: String,
seed_bytes: Vec<u8>,
era: Vec<u8>,
day_bytes: Vec<u8>,
log2: u32,
/// The worker's reference: the pack's program and day dataset, checked against the pack's vectors
reference: Epoch,
}
/// The pack as the worker sees it: program.json's program (generator, id and class as recorded) over the pack's
/// day dataset, trusted only after its 96 vectors reproduce.
fn load(rel: &str, want: ProgramClass) -> PackRef {
let dir = pack_dir(rel);
let j = json(&dir, "program.json");
let class = match j["program_class"].as_str().unwrap() {
"v3" => ProgramClass::V3,
"v4" => ProgramClass::V4,
other => panic!("{rel}: class {other}"),
};
assert_eq!(class, want, "{rel}: the pack's class");
let seed_label = j["seed"].as_str().unwrap().to_string();
let seed_bytes = unhex(j["seed_bytes"].as_str().unwrap()).unwrap();
let era = unhex(j["era_seed_bytes"].as_str().unwrap()).unwrap();
let day_bytes = unhex(j["dataset"]["day_bytes"].as_str().unwrap()).unwrap();
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
let program = igneum_pow::generate_from_seed_bytes_program_class(&seed_label, &seed_bytes, class, Some(&era));
assert_eq!(program.generator, j["generator"].as_u64().unwrap() as u32, "{rel}: generator");
assert_eq!(program.generator, class.generator_version(), "{rel}: the generator is the class's");
assert_eq!(program.program_id(), hex64(&j["program_id"]), "{rel}: program id");
let shape = Shape::for_class(&program.class);
let mut dataset = DatasetSource::from_key_shape(igneum_pow::seed::seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2, shape);
dataset.key_bytes = day_bytes.clone();
let reference = Epoch { program, dataset };
let v = json(&dir, "vectors.json");
let warps = v["warps"].as_array().unwrap();
assert_eq!(warps.len(), 3, "{rel}: three vector warps");
for w in warps {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let got = reference.hash_warp(base);
for (lane, e) in w["expected"].as_array().unwrap().iter().enumerate() {
assert_eq!(got[lane], hex64(e), "{rel}: vector base {base} lane {lane}");
}
}
PackRef { class, seed_label, seed_bytes, era, day_bytes, log2, reference }
}
/// The re-check's epoch: the chain seam the node engine and the miner call, from the template's four values.
/// The packs are day-0 caches, so days since genesis is 0 and the genesis dataset size is the pack's.
fn recheck_epoch(p: &PackRef) -> Epoch {
Epoch { program: Epoch::chain_program(&p.seed_bytes, Some(&p.era), p.class, &p.seed_label), dataset: Epoch::chain_dataset_day(&p.day_bytes, p.class, 0, p.log2) }
}
const PREHASH: [u8; 32] = [0x5a; 32];
const NONCE: u64 = 0x1234_5678_0000_0bb7; // high word 0x12345678, lane 0x0bb7 (warp base 0x0ba0, lane 23)
/// The bound lane hash of one nonce on an epoch, the way a worker computes it (init words from the prehash and the
/// nonce's high word, the warp at the lane's base) and the way `EpochRef::hash_bound` re-checks it.
fn bound(e: &Epoch, prehash: &[u8; 32], nonce: u64) -> u64 {
let init = block_init_words(prehash, nonce);
let lane = lane_nonce(nonce);
interpret_warp_init(&e.program, &init, lane & !31, &e.dataset).hashes[(lane & 31) as usize]
}
#[test]
fn cpu_recheck_equals_the_worker_reference_for_class_v3_and_class_v4() {
let v3 = load("packs-ca2-mixer/mx8-devnet-epoch0", ProgramClass::V3);
let v4 = load("packs-ca3-v4/v4-devnet-epoch0", ProgramClass::V4);
assert_eq!(v3.seed_bytes, v4.seed_bytes, "the two packs share the epoch seed (devnet genesis), so only the class differs");
assert_eq!(v3.era, v4.era);
let mut hashes = Vec::new();
for p in [&v3, &v4] {
let re = recheck_epoch(p);
assert_eq!(re.program.program_id(), p.reference.program.program_id(), "{:?}: the seam's program is the pack's", p.class);
assert_eq!(re.program.generator, p.reference.program.generator);
assert_eq!(re.program.instrs.len(), p.reference.program.instrs.len());
let worker = bound(&p.reference, &PREHASH, NONCE);
let cpu = bound(&re, &PREHASH, NONCE);
assert_eq!(cpu, worker, "{:?}: CPU re-check {cpu:016x} against the worker's reference {worker:016x}", p.class);
// the same nonce through a second prehash, so the agreement is not one lucky lane
let (w2, c2) = (bound(&p.reference, &[0xa5; 32], NONCE ^ 0x1f), bound(&re, &[0xa5; 32], NONCE ^ 0x1f));
assert_eq!(c2, w2, "{:?}: second nonce", p.class);
hashes.push(cpu);
}
assert_ne!(hashes[0], hashes[1], "class v3 and class v4 programs of one seed hash a nonce differently");
// the fleet's mismatch: a fixed class v2 program on the class v3 epoch's seed
let v2 = Epoch { program: Epoch::chain_program(&v3.seed_bytes, None, ProgramClass::V2, &v3.seed_label), dataset: Epoch::chain_dataset_day(&v3.day_bytes, ProgramClass::V2, 0, v3.log2) };
assert_ne!(bound(&v2, &PREHASH, NONCE), hashes[0], "a class v2 re-check refuses a class v3 share");
assert_ne!(bound(&v2, &PREHASH, NONCE), hashes[1], "a class v2 re-check refuses a class v4 share");
}
/// The generator-4 program id rule (counter-asic-3-status.md, the blocker closed 6 October 2026): the v4 program of
/// a seed carries another id than the v3 program of the same seed, so a stale worker across the activation sees the
/// mismatch, while v2 and v3 ids stay as the packs pinned them.
#[test]
fn program_ids_differ_between_class_v3_and_class_v4_of_one_seed() {
let v3 = load("packs-ca2-mixer/mx8-devnet-epoch0", ProgramClass::V3);
let v4 = load("packs-ca3-v4/v4-devnet-epoch0", ProgramClass::V4);
assert_eq!(v3.reference.program.program_id(), 0x73bc_bfe8_ccf9_88f1, "the v3 control's id as pinned");
// the amended class v4 (AP-F8-1, AP-F8-3): sub-version 3 (object byte 7; the acceptance executes the shadow block)
// is the pinned id below; c120d7963abdcd96 (the 6 October stream, byte 4), 1a4230699a6b9c60 (sub-version 1, the
// one-writer rule, 0.3.20's and 0.3.21's byte 5) and a788661687db4bb3 (sub-version 2, never shipped) must differ
assert_ne!(v4.reference.program.program_id(), 0xc120_d796_3abd_cd96, "the pre-amendment v4 id must differ");
assert_ne!(v4.reference.program.program_id(), 0x1a42_3069_9a6b_9c60, "the sub-version-1 id must differ");
assert_ne!(v4.reference.program.program_id(), 0xa788_6616_87db_4bb3, "the sub-version-2 id must differ");
assert_eq!(v4.reference.program.program_id(), 0xa785_0016_87d8_688a, "the sub-version-3 v4 id as pinned");
assert_ne!(v3.reference.program.program_id(), v4.reference.program.program_id());
}

View file

@ -0,0 +1,769 @@
//! Soundness tests of layer 3 of `docs/plans/counter-asic-2.md`: the per-warp scratch with read-modify-writes
//! (variant 5 of the read-width experiment, `LoadClass::scratch(k, kb)`). Analysis and results:
//! `docs/analysis/scratch-soundness.md`. Every test is parametric over the class's slot count
//! (`scratch_slots_per_lane()`), so the 32 and 128 KiB geometries and any later one run the same checks.
//!
//! What runs under plain `cargo test`:
//! 1. `rewrite_is_a_bijection_of_the_fold_value`, `fill_is_a_bijection_of_the_nonce`: the written words as
//! functions (question 1).
//! 2. `written_words_unbiased_and_rehit_rates`: bit bias of every written word over 2^11 units x 3 seeds per class
//! (the TESTS.md section 3 shape), and the measured slot re-hit rate against the birthday formula (question 2).
//! 3. `edge_programs_match_the_hand_model`: hand-built programs that drive every read-modify-write of a hash to
//! slot 0, slot MASK, through out-of-range registers, to one slot per lane, alternating two slots, and 16
//! read-modify-writes per iteration on one slot; the interpreter against an independent hand model, and the
//! hand model shown to have teeth (question 3, CPU half).
//! 4. `scr_packs_regenerate_and_pass_the_static_scratch_check`: every emitted kernel of every scr pack under
//! `proto-cuda/packs-readwidth` regenerates from its program.json and passes the static scratch-mask check;
//! the check is shown to fail on four deliberate breaks (question 4).
//! 5. `fuzz_scr_programs_cpu`: 200 generated scratch programs over the six classes, generator contract on every
//! instruction, 4 units each at base nonces across the 32-bit range including the wrap; with
//! `IGNEUM_SCRATCH_PACKS_OUT=<dir>` it also writes the packs (and the edge packs) for the Metal runs of
//! `proto-metal/packbench` (question 3 GPU half, question 4, `TESTS.md` section 9 shape).
use igneum_pow::emit::{
cuda_kernel, cuda_kernel_bound, export_pack, metal_program, metal_program_bound, opencl_kernel,
opencl_kernel_bound, vectors_json, LoadSource,
};
use igneum_pow::generator::{
generate_class, generate_from_seed_bytes_class, Instr, LoadClass, Op, Program, GENERATOR_VERSION, INSTR_COUNT,
ITERATIONS, LANES,
};
use igneum_pow::seed::{seed_words_from_bytes, SplitMix64};
use igneum_pow::verify::{
fold_words, interpret_warp_scratch, scratch_fill, scratch_rewrite, splitmix32, DatasetMode, DatasetSource,
Epoch, ScratchEvent, FOLD_MUL, FOLD_ROT,
};
use serde_json::Value;
use std::collections::HashMap;
use std::path::PathBuf;
/// The classes under study: the two capped geometries (32 and 128 KiB per warp: 64 and 256 slots per lane) at the
/// RMW shares the readwidth branch measures.
const CLASSES: [&str; 6] = ["scr2k32", "scr4k32", "scr8k32", "scr2k128", "scr4k128", "scr8k128"];
fn class(name: &str) -> LoadClass {
LoadClass::parse(name).unwrap_or_else(|| panic!("class {name}"))
}
// ---------------------------------------------------------------------------------------------------------------
// 1. The written words as functions (question 1)
// ---------------------------------------------------------------------------------------------------------------
/// For a fixed slot content `w`, each of the three rewritten words is a bijection of the fold value `x`
/// (`x ^ w1`, `rotl(x, 7) ^ w2`, `x + w0`), so the rewrite is injective in `x` and a uniform `x` gives a uniform
/// word in every position. Checked over 2^16 consecutive `x` for 16 random `w`.
#[test]
fn rewrite_is_a_bijection_of_the_fold_value() {
let mut rng = SplitMix64::new(0x7363_7261_7463_6801);
for _ in 0..16 {
let w = [rng.next() as u32, rng.next() as u32, rng.next() as u32];
let x0 = rng.next() as u32;
let mut seen = [vec![false; 1 << 16], vec![false; 1 << 16], vec![false; 1 << 16]];
for i in 0..(1u32 << 16) {
let x = x0.wrapping_add(i);
let out = scratch_rewrite(x, &w);
for j in 0..3 {
// a bijection of x maps 2^16 consecutive x to 2^16 distinct words; the low 16 bits alone are
// distinct for the xor words (x ^ c) and for the add word (x + c), since both act on the low 16
// bits as bijections of the low 16 bits of x; the rotl word is checked on its rotated-back bits
let key = if j == 1 { out[j].rotate_right(7) & 0xffff } else { out[j] & 0xffff };
assert!(!seen[j][key as usize], "word {j} repeats inside 2^16 consecutive x");
seen[j][key as usize] = true;
}
}
}
// The rewrite inverts: from the old content and any ONE written word the fold value is recovered, so a
// rewritten slot carries exactly 32 bits of new state (the point of question 2's arithmetic).
let w = [0x1234_5678, 0x9abc_def0, 0x0fed_cba9];
let x = 0xdead_beef;
let out = scratch_rewrite(x, &w);
assert_eq!(out[0] ^ w[1], x);
assert_eq!((out[1] ^ w[2]).rotate_right(7), x);
assert_eq!(out[2].wrapping_sub(w[0]), x);
}
/// For a fixed (seed, slot, j) the fill is a bijection of the lane nonce: `splitmix32` is a bijection of its
/// 32-bit input and the input `((base + lane) ^ s) + c` is a bijection of `base + lane`. Over 2^16 consecutive
/// nonces no fill word repeats, for 8 slots x 3 words.
#[test]
fn fill_is_a_bijection_of_the_nonce() {
let seed = seed_words_from_bytes(b"igneum-genesis");
for slot in [0u32, 1, 63, 64, 255, 1023, 2047] {
for j in 0..3u32 {
let mut words: Vec<u32> = (0..(1u32 << 16)).map(|n| scratch_fill(&seed, n, 0, slot, j)).collect();
words.sort_unstable();
words.dedup();
assert_eq!(words.len(), 1 << 16, "slot {slot} word {j}: fill words of 2^16 consecutive nonces are distinct");
}
}
// base + lane is the lane nonce: the fill of lane l at base b is the fill of lane 0 at base b + l
assert_eq!(scratch_fill(&seed, 0x1000, 7, 5, 2), scratch_fill(&seed, 0x1007, 0, 5, 2));
// and it wraps with the nonce: base 0xffffffe0, lane 31 is nonce 0xffffffff; lane 32 would be nonce 0
assert_eq!(scratch_fill(&seed, 0xffff_ffe0, 32, 5, 2), scratch_fill(&seed, 0, 0, 5, 2));
// the three word positions of one slot and nonce are three different permutation outputs
let f: Vec<u32> = (0..3).map(|j| scratch_fill(&seed, 12345, 7, 17, j)).collect();
assert!(f[0] != f[1] && f[1] != f[2] && f[0] != f[2]);
}
// ---------------------------------------------------------------------------------------------------------------
// 2. Uniformity of the written words and the slot re-hit rate (questions 1 and 2)
// ---------------------------------------------------------------------------------------------------------------
/// Birthday arithmetic: the expected number of distinct slots after `n` uniform draws from `s` slots.
fn expected_distinct(s: usize, n: usize) -> f64 {
let s = s as f64;
s * (1.0 - (1.0 - 1.0 / s).powi(n as i32))
}
struct ClassStats {
units: usize,
events: usize,
hits: usize,
/// ones count per bit of the written words, 3 x 32
ones: [[u64; 32]; 3],
/// ones count per bit of written XOR read (the change the rewrite makes to the slot)
delta_ones: [[u64; 32]; 3],
/// re-hit depth histogram: how many earlier RMWs the slot had seen in this unit (0 = first touch)
depth: Vec<usize>,
max_depth: usize,
/// how often each slot index was addressed (the slot comes from a register's low bits)
slot_hist: Vec<u64>,
}
fn class_stats(name: &str, seeds: &[&str], units_per_seed: usize) -> ClassStats {
let c = class(name);
let mut st = ClassStats {
units: 0,
events: 0,
hits: 0,
ones: [[0; 32]; 3],
delta_ones: [[0; 32]; 3],
depth: vec![0; 256],
max_depth: 0,
slot_hist: vec![0; c.scratch_slots_per_lane()],
};
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 28);
for seed in seeds {
let p = generate_class(seed, c);
assert_eq!(p.scratch_ops_per_hash(), c.scratch_slots() * ITERATIONS);
for u in 0..units_per_seed {
let base = (u as u32).wrapping_mul(32).wrapping_add(0x4000_0000);
let (_, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true);
assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES);
let mut count: HashMap<(u8, u32), usize> = HashMap::new();
for e in &ev {
assert!(e.slot < c.scratch_slots_per_lane() as u32, "slot inside the lane's scratch");
let d = count.entry((e.lane, e.slot)).or_insert(0);
assert_eq!(e.hit, *d > 0, "hit flag agrees with the unit's own history");
assert_eq!(e.written, scratch_rewrite(e.x, &e.read));
if !e.hit {
let fill = [
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 0),
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 1),
scratch_fill(&p.seed, base, e.lane as u32, e.slot, 2),
];
assert_eq!(e.read, fill, "a first touch reads the fill");
}
st.depth[(*d).min(255)] += 1;
st.max_depth = st.max_depth.max(*d);
st.slot_hist[e.slot as usize] += 1;
*d += 1;
st.events += 1;
st.hits += e.hit as usize;
for j in 0..3 {
for b in 0..32 {
st.ones[j][b] += ((e.written[j] >> b) & 1) as u64;
st.delta_ones[j][b] += (((e.written[j] ^ e.read[j]) >> b) & 1) as u64;
}
}
}
st.units += 1;
}
}
st
}
/// Bit bias of every written word (and of the change each rewrite makes) within 6 sigma of a fair coin, over
/// 3 seeds x 2^11 units per class (131,072 hashes per seed set); the slot re-hit rate against the birthday
/// formula within 3 percent relative. The table printed here is the one in the analysis.
#[test]
fn written_words_unbiased_and_rehit_rates() {
let seeds = ["igneum-genesis", "igneum-genesis/stats1", "igneum-genesis/stats2"];
let units = 1usize << 11;
println!("class | slots/lane | RMW/hash | events | re-hits | re-hit % | birthday % | slot chi2 z (spread) | max depth | max bias sigma | max delta bias sigma");
for name in CLASSES {
let c = class(name);
let st = class_stats(name, &seeds, units);
let n = st.events as f64;
let sigma = (n / 4.0).sqrt();
let mut worst = 0.0f64;
let mut worst_delta = 0.0f64;
for j in 0..3 {
for b in 0..32 {
let z = (st.ones[j][b] as f64 - n / 2.0).abs() / sigma;
let zd = (st.delta_ones[j][b] as f64 - n / 2.0).abs() / sigma;
assert!(z <= 6.0, "{name}: written word {j} bit {b} biased: {z:.2} sigma");
assert!(zd <= 6.0, "{name}: rewrite delta word {j} bit {b} biased: {zd:.2} sigma");
worst = worst.max(z);
worst_delta = worst_delta.max(zd);
}
}
let per_lane_hash = c.scratch_slots() * ITERATIONS;
let s = c.scratch_slots_per_lane();
let exp_hits = per_lane_hash as f64 - expected_distinct(s, per_lane_hash);
let exp_pct = 100.0 * exp_hits / per_lane_hash as f64;
let got_pct = 100.0 * st.hits as f64 / st.events as f64;
// chi-square of the slot histogram against uniform (df = s - 1): the slot is a register's low bits, and
// the measured re-hit rate runs above the uniform birthday rate (the finding of the analysis, question 2)
let expect_per_slot = n / s as f64;
let chi2: f64 = st.slot_hist.iter().map(|&h| (h as f64 - expect_per_slot).powi(2) / expect_per_slot).sum();
let chi2_z = (chi2 - (s as f64 - 1.0)) / (2.0 * (s as f64 - 1.0)).sqrt();
let hot = *st.slot_hist.iter().max().unwrap() as f64 / expect_per_slot;
let cold = *st.slot_hist.iter().min().unwrap() as f64 / expect_per_slot;
println!(
"{name} | {s} | {per_lane_hash} | {} | {} | {got_pct:.2} | {exp_pct:.2} | {chi2_z:.1} (hottest slot {hot:.2}x, coldest {cold:.2}x) | {} | {worst:.2} | {worst_delta:.2}",
st.events, st.hits, st.max_depth
);
// a regression band, not a uniformity claim: the rate sits between the uniform birthday rate and twice it
assert!(
got_pct >= 0.9 * exp_pct && got_pct <= 2.0 * exp_pct,
"{name}: re-hit rate {got_pct:.2}% against birthday {exp_pct:.2}%"
);
// depth histogram: the number of earlier RMWs a re-hit slot had seen in the unit
let shown: Vec<String> = st.depth.iter().take(st.max_depth + 1).enumerate().map(|(d, n)| format!("{d}:{n}")).collect();
println!(" depth histogram {}", shown.join(" "));
}
}
// ---------------------------------------------------------------------------------------------------------------
// 3. Hand-built edge programs against an independent hand model (question 3, CPU half)
// ---------------------------------------------------------------------------------------------------------------
fn ins(op: Op, dst: u8, src: u8) -> Instr {
Instr { op, dst, src, src2: 0, imm: 0, imm2: 0, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 }
}
fn add_imm(dst: u8, src: u8, imm: u32) -> Instr {
Instr { op: Op::Add, dst, src, src2: 0, imm, imm2: imm, rot: 1, bit: 0, mask: 1, width: 1, win: 0, off: 0 }
}
/// A hand-built program of class `c` named `name` (its seed is the name, so its fill words and init words are
/// its own). These bypass the generator and the acceptance rule, like `TESTS.md` section 2; `sub r, r` zeroes a
/// register as the Swift edge set does.
fn edge(name: &str, c: LoadClass, instrs: Vec<Instr>) -> Program {
let seed_string = format!("igneum-scratch-edge/{name}");
let seed_bytes = seed_string.as_bytes().to_vec();
let k = instrs.iter().filter(|i| i.op == Op::Scratch).count();
assert_eq!(k, c.scratch_slots(), "{name}: the class carries the program's scratch count");
Program {
seed: seed_words_from_bytes(&seed_bytes),
seed_string,
seed_bytes,
generator: GENERATOR_VERSION,
attempt: 0,
class: c,
era_bytes: None,
instrs,
shadow: Vec::new(),
}
}
/// The edge set for a scratch of `kb` KiB per warp. Each entry: (name, what it drives, program).
fn edge_programs(kb: u8) -> Vec<(String, &'static str, Program)> {
let m = LoadClass::scratch(1, kb).scratch_slot_mask();
let dsts = [2u8, 3, 4, 5, 6, 7, 0, 2, 3, 4, 5, 6, 7, 0, 2, 3];
let scr = |n: usize, src: u8| -> Vec<Instr> { (0..n).map(|i| ins(Op::Scratch, dsts[i], src)).collect() };
let mut v = Vec::new();
// every RMW of the hash to slot 0 through a zero register: 64 dependent RMWs on one slot per lane
let mut p = vec![ins(Op::Sub, 1, 1)];
p.extend(scr(8, 1));
v.push(("slot0".to_string(), "r1 = 0: every RMW to slot 0", edge(&format!("slot0/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot MASK through the in-range register MASK
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m)];
p.extend(scr(8, 1));
v.push(("slotmask".to_string(), "r1 = MASK: every RMW to the last slot", edge(&format!("slotmask/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot MASK through the out-of-range register 0xffffffff
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, 1), ins(Op::Sub, 1, 2)];
p.extend(scr(8, 1));
v.push(("ones".to_string(), "r1 = 0xffffffff: masked to the last slot", edge(&format!("ones/k{kb}"), LoadClass::scratch(8, kb), p)));
// slot 0 through the out-of-range register MASK + 1
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(1, 2, m.wrapping_add(1))];
p.extend(scr(8, 1));
v.push(("maskplus1".to_string(), "r1 = MASK + 1: masked to slot 0", edge(&format!("maskplus1/k{kb}"), LoadClass::scratch(8, kb), p)));
// 16 RMWs per iteration on slot 0: 128 dependent RMWs on one slot per lane per hash
let mut p = vec![ins(Op::Sub, 1, 1)];
p.extend(scr(16, 1));
v.push(("sixteen".to_string(), "16 RMWs per iteration on slot 0", edge(&format!("sixteen/k{kb}"), LoadClass::scratch(16, kb), p)));
// one slot per lane from the init words: lanes with equal slots would show any cross-lane aliasing
// (r5 is the slot register and is never a destination here)
let p: Vec<Instr> = [0u8, 1, 2, 3, 4, 6, 7, 0].iter().map(|&d| ins(Op::Scratch, d, 5)).collect();
v.push(("lanevar".to_string(), "r5 never written: one init-dependent slot per lane", edge(&format!("lanevar/k{kb}"), LoadClass::scratch(8, kb), p)));
// alternating slot 0 and slot MASK inside one iteration
let mut p = vec![ins(Op::Sub, 1, 1), ins(Op::Sub, 2, 2), add_imm(2, 1, m)];
for (i, &d) in [3u8, 4, 5, 6, 7, 0, 3, 4].iter().enumerate() {
// r1 and r2 hold the two slots and are never destinations
p.push(ins(Op::Scratch, d, if i % 2 == 0 { 1 } else { 2 }));
}
v.push(("twoslots".to_string(), "slot 0 and slot MASK alternating", edge(&format!("twoslots/k{kb}"), LoadClass::scratch(8, kb), p)));
v
}
/// The hand model: a second, minimal interpreter for the ops the edge programs use (sub, add, scratch), with its
/// own slot store keyed by (lane, slot). `mutate` swaps the rewrite's words to show the comparison has teeth.
fn hand_model(p: &Program, base: u32, mutate: bool) -> [u64; 32] {
let seed = &p.seed;
let m = p.class.scratch_slot_mask();
let mut r = [[0u32; LANES]; 8];
for lane in 0..LANES {
let nonce = base.wrapping_add(lane as u32);
for i in 0..8 {
let mut x = nonce ^ seed[i];
x = x.wrapping_add(0x9e3779b9u32.wrapping_mul(i as u32 + 1));
x = splitmix32(x);
r[i][lane] = x ^ seed[(i + 1) & 7];
}
}
let mut store: HashMap<(usize, u32), [u32; 3]> = HashMap::new();
for _ in 0..ITERATIONS {
let sel = r[0];
for ins in &p.instrs {
let (d, a) = (ins.dst as usize, ins.src as usize);
match ins.op {
Op::Sub => {
for lane in 0..LANES {
r[d][lane] = r[d][lane].wrapping_sub(r[a][lane]);
}
}
Op::Add => {
for lane in 0..LANES {
let c = if (sel[lane] >> ins.bit) & 1 != 0 { ins.imm2 } else { ins.imm };
r[d][lane] = r[d][lane].wrapping_add(r[a][lane]).wrapping_add(c);
}
}
Op::Scratch => {
for lane in 0..LANES {
let slot = r[a][lane] & m;
let w = *store.entry((lane, slot)).or_insert_with(|| {
let mut f = [0u32; 3];
for j in 0..3u32 {
// the fill, written out in full rather than through verify::scratch_fill
let n = base.wrapping_add(lane as u32);
f[j as usize] = splitmix32(
(n ^ seed[j as usize])
.wrapping_add(slot.wrapping_mul(0x9E37_79B1))
.wrapping_add((j + 1).wrapping_mul(0x85EB_CA77)),
);
}
f
});
let mut x = r[d][lane] ^ w[0];
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[1];
x = x.rotate_left(FOLD_ROT).wrapping_mul(FOLD_MUL) ^ w[2];
r[d][lane] = x;
let out = if mutate {
[x.rotate_left(7) ^ w[2], x ^ w[1], x.wrapping_add(w[0])]
} else {
[x ^ w[1], x.rotate_left(7) ^ w[2], x.wrapping_add(w[0])]
};
store.insert((lane, slot), out);
}
}
other => panic!("the hand model does not implement {other:?}"),
}
}
}
let mut out = [0u64; 32];
for lane in 0..LANES {
let lo = r[0][lane] ^ r[1][lane].rotate_left(7) ^ r[2][lane].rotate_left(14) ^ r[3][lane].rotate_left(21);
let hi = r[4][lane] ^ r[5][lane].rotate_left(9) ^ r[6][lane].rotate_left(18) ^ r[7][lane].rotate_left(27);
out[lane] = ((hi as u64) << 32) | lo as u64;
}
out
}
/// The four unit bases of every edge vector: 0 and 32 (two consecutive units, the pair a one-warp persistent
/// launch runs on one arena), a unit straddling 2^31, and the unit that wraps past 2^32.
const EDGE_BASES: [u32; 4] = [0, 32, 0x7fff_fff0, 0xffff_ffe0];
#[test]
fn edge_programs_match_the_hand_model() {
let ds = DatasetSource::new("2026-10-03", DatasetMode::ClosedForm, 24);
let mut cases = 0;
for kb in [32u8, 128] {
for (name, what, p) in edge_programs(kb) {
let slots = p.class.scratch_slots_per_lane();
for base in EDGE_BASES {
let (res, ev) = interpret_warp_scratch(&p, &p.seed, base, &ds, true);
let hand = hand_model(&p, base, false);
assert_eq!(res.hashes, hand, "{name} k{kb} base {base:#x}: interpreter against the hand model ({what})");
assert_ne!(res.hashes, hand_model(&p, base, true), "{name} k{kb}: the comparison has teeth");
// the slots the trace saw are the ones the program was built to drive
let slot_set: std::collections::BTreeSet<u32> = ev.iter().map(|e| e.slot).collect();
let m = (slots - 1) as u32;
match name.as_str() {
"slot0" | "maskplus1" | "sixteen" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![0]),
"slotmask" | "ones" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![m]),
"twoslots" => assert_eq!(slot_set.into_iter().collect::<Vec<_>>(), vec![0, m]),
"lanevar" => {
for e in &ev {
assert!(e.slot <= m);
}
}
_ => unreachable!(),
}
// the chain depth on the driven slot: every RMW after the first per lane is a re-hit
let per_lane = p.scratch_ops_per_hash();
let hits = ev.iter().filter(|e| e.hit).count();
let expected_hits = match name.as_str() {
"twoslots" => (per_lane - 2) * LANES,
_ => (per_lane - 1) * LANES,
};
assert_eq!(hits, expected_hits, "{name} k{kb}: re-hits");
cases += 1;
}
}
}
assert_eq!(cases, 2 * 7 * 4);
}
// ---------------------------------------------------------------------------------------------------------------
// 4. The static scratch check over every emitted kernel of every scr pack (question 4)
// ---------------------------------------------------------------------------------------------------------------
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum Dialect {
Metal,
Cuda,
OpenCl,
}
/// The static scratch check: every scratch read-modify-write in an emitted kernel has the one masked form the
/// emitter writes, the arena is the lane's own `slots x 4` words, the tag is `salt + unit`, and nothing else
/// touches the scratch. Like the dataset mask check of `TESTS.md` section 5 and `tests/packs.rs`, a text check:
/// the guarantee is that the emitter has one template and it masks.
pub fn scratch_text_check(text: &str, dialect: Dialect, k: usize, slots: usize, kernels: usize) -> Result<(), String> {
assert!(kernels >= 1);
// every count below is per hash kernel; an OpenCL bound file carries igneum_hash and igneum_hash_bound
let k = k * kernels;
assert!(slots.is_power_of_two() && slots >= 1);
let mask = (slots - 1) as u32;
let wpl = slots * 4;
let (u, load, store, ptr) = match dialect {
Dialect::Metal => ("uint", "uint4 v_ = *(device const uint4*)(arena + s_ * 4u);", "*(device uint4*)(arena + s_ * 4u) = uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "device uint* arena"),
Dialect::Cuda => ("uint32_t", "uint4 v_ = *(const uint4*)(arena + s_ * 4u);", "*(uint4*)(arena + s_ * 4u) = make_uint4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_); }", "uint32_t* arena"),
Dialect::OpenCl => ("uint", "uint4 v_ = vload4(s_, arena);", "vstore4(IGNEUM_U4(tag, x_ ^ w1_, rotl_imm(x_, 7u) ^ w2_, x_ + w0_), s_, arena); }", "__global uint* arena"),
};
let count = |needle: &str| text.matches(needle).count();
let mut errs = Vec::new();
let mut expect = |what: &str, got: usize, want: usize| {
if got != want {
errs.push(format!("{what}: {got}, expected {want}"));
}
};
// k slot computations, each masked with exactly the class's mask and immediately followed by the one load form
expect("slot definitions `{ u s_ = r`", count(&format!("{{ {u} s_ = r")), k);
expect("masked slot followed by the load", count(&format!(" & {mask}u; {load}")), k);
expect("stores of the tagged slot", count(store), k);
expect("tag compares", count("(v_.x == tag)"), k);
expect("fill calls (three per RMW)", count("scr_fill(gbase, lane, s_, "), 3 * k);
// the arena: one definition with the class's words per lane, and 2k uses (one load, one store per RMW)
expect("arena definition", count(&format!("{ptr} = scratch + ((size_t)warp_ * 32u + lane) * {wpl}u;")), kernels);
expect("arena mentions (definition + load + store per RMW)", count("arena"), kernels + 2 * k);
expect("tag definition `tag = salt + g_`", count(&format!("{u} tag = salt + g_;")), kernels);
expect("direct scratch indexing", count("scratch["), 0);
expect("scratch pointer arithmetic outside the arena definition", count("scratch +"), kernels);
// no other mask value on a slot: every `s_ = r` line carries the class mask and nothing else carries ` & Nu; uint4 v_`
let any_mask_load = count(&format!("u; {load}"));
expect("loads preceded by some mask (must all be the class mask)", any_mask_load, k);
if errs.is_empty() {
Ok(())
} else {
Err(errs.join("; "))
}
}
fn packs_rw_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../proto-cuda/packs-readwidth")
}
fn scr_packs() -> Vec<String> {
let mut v: Vec<String> = std::fs::read_dir(packs_rw_dir())
.unwrap()
.map(|d| d.unwrap().file_name().to_string_lossy().to_string())
.filter(|n| n.starts_with("scr"))
.collect();
v.sort();
v
}
fn read_pack(pack: &str, file: &str) -> String {
let p = packs_rw_dir().join(pack).join(file);
std::fs::read_to_string(&p).unwrap_or_else(|e| panic!("read {}: {e}", p.display()))
}
/// Every scr pack regenerates from its program.json (seed bytes, class, day bytes, size) to the same six kernel
/// texts, byte for byte, and every one of those texts passes the static scratch check for the class's k and slot
/// count; the check fails on four deliberate breaks of a copy of the Metal text (mask dropped, mask changed, arena
/// stride changed, a stray scratch access) and on the OpenCL and CUDA twins of the first.
#[test]
fn scr_packs_regenerate_and_pass_the_static_scratch_check() {
let packs = scr_packs();
assert!(packs.len() >= 6, "the scr packs: {packs:?}");
let mut checked = 0;
let mut sample_metal = String::new();
let mut sample_cl = String::new();
let mut sample_cu = String::new();
let mut sample_k = 0;
let mut sample_slots = 0;
for pack in &packs {
let j: Value = serde_json::from_str(&read_pack(pack, "program.json")).unwrap();
let name = j["load_class"].as_str().unwrap();
let c = class(name);
assert_eq!(&format!("{name}"), pack, "pack directory named after its class");
let seed = j["seed"].as_str().unwrap();
let seed_bytes = igneum_pow::bind::unhex(j["seed_bytes"].as_str().unwrap()).unwrap();
let day_bytes = igneum_pow::bind::unhex(j["dataset"]["day_bytes"].as_str().unwrap()).unwrap();
let log2 = j["dataset"]["log2_words"].as_u64().unwrap() as u32;
assert_eq!(j["dataset_mode"].as_str().unwrap(), "memory-hard");
let program = generate_from_seed_bytes_class(seed, &seed_bytes, c);
assert_eq!(program.class, c);
assert_eq!(program.program_id(), u64::from_str_radix(j["program_id"].as_str().unwrap().trim_start_matches("0x"), 16).unwrap());
let mut dataset = DatasetSource::from_key(seed_words_from_bytes(&day_bytes), DatasetMode::MemoryHard, log2);
dataset.key_bytes = day_bytes;
let e = Epoch { program, dataset };
let p = &e.program;
let mp = e.dataset.memhard().map(|m| &m.params);
let k = c.scratch_slots();
let slots = c.scratch_slots_per_lane();
assert_eq!(p.scratch_ops_per_hash(), k * ITERATIONS);
for (file, text, dialect, kernels) in [
("program.metal", metal_program(p, log2, LoadSource::Stored), Dialect::Metal, 1),
("program_bound.metal", metal_program_bound(p, log2), Dialect::Metal, 1),
("kernel.cu", cuda_kernel(p, mp), Dialect::Cuda, 1),
("kernel_bound.cu", cuda_kernel_bound(p, mp), Dialect::Cuda, 1),
("kernel.cl", opencl_kernel(p, mp), Dialect::OpenCl, 1),
// the OpenCL bound file carries igneum_hash and igneum_hash_bound
("kernel_bound.cl", opencl_kernel_bound(p, mp), Dialect::OpenCl, 2),
] {
let on_disk = read_pack(pack, file);
assert_eq!(on_disk, text, "{pack}/{file}: the pack is the emitter's text");
// scr0 is the persistent control: an arena and a tag, no read-modify-write; the check holds with k = 0
scratch_text_check(&on_disk, dialect, k, slots, kernels).unwrap_or_else(|e| panic!("{pack}/{file}: {e}"));
checked += 1;
}
// the vectors of the pack are the CPU's
let v: Value = serde_json::from_str(&read_pack(pack, "vectors.json")).unwrap();
for w in v["warps"].as_array().unwrap() {
let base = w["base_nonce"].as_u64().unwrap() as u32;
let got = e.hash_warp(base);
for (lane, x) in w["expected"].as_array().unwrap().iter().enumerate() {
let want = u64::from_str_radix(x.as_str().unwrap().trim_start_matches("0x"), 16).unwrap();
assert_eq!(got[lane], want, "{pack}: base {base} lane {lane}");
}
}
if k == 4 && slots == 64 {
sample_metal = read_pack(pack, "program.metal");
sample_cl = read_pack(pack, "kernel.cl");
sample_cu = read_pack(pack, "kernel.cu");
sample_k = k;
sample_slots = slots;
}
}
assert_eq!(checked, packs.len() * 6);
println!("static scratch check: {checked} kernels over {} scr packs", packs.len());
// The deliberate breaks (the watcher rule of CLAUDE.md: a check is trusted once it fails on a known-broken
// case). Each must be caught; the message names what.
assert!(sample_k == 4 && sample_slots == 64, "scr4k32 is in the pack set");
let mask = format!(" & {}u; uint4 v_", sample_slots - 1);
let broken_mask = sample_metal.replacen(&mask, "; uint4 v_", 1);
assert_ne!(broken_mask, sample_metal);
let e = scratch_text_check(&broken_mask, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 1 (one mask dropped, Metal): {e}");
let wrong_mask = sample_metal.replace(" & 63u;", " & 127u;");
let e = scratch_text_check(&wrong_mask, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 0, expected 4"), "{e}");
println!("break 2 (mask 63 -> 127 on every RMW, Metal): {e}");
let wrong_stride = sample_metal.replace("* 256u;", "* 128u;");
let e = scratch_text_check(&wrong_stride, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("arena definition: 0, expected 1"), "{e}");
println!("break 3 (arena stride 256 -> 128 words, Metal): {e}");
let stray = format!("{sample_metal}\n// stray\n// arena[0] = 0u; scratch[1] = 1u;\n");
let e = scratch_text_check(&stray, Dialect::Metal, 4, 64, 1).unwrap_err();
assert!(e.contains("arena mentions") && e.contains("direct scratch indexing: 1, expected 0"), "{e}");
println!("break 4 (a stray arena and scratch access, Metal): {e}");
let e = scratch_text_check(&sample_cl.replacen(" & 63u; uint4 v_ = vload4", "; uint4 v_ = vload4", 1), Dialect::OpenCl, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 5 (one mask dropped, OpenCL): {e}");
let e = scratch_text_check(&sample_cu.replacen(" & 63u; uint4 v_ = *(const uint4*)", "; uint4 v_ = *(const uint4*)", 1), Dialect::Cuda, 4, 64, 1).unwrap_err();
assert!(e.contains("masked slot followed by the load: 3, expected 4"), "{e}");
println!("break 6 (one mask dropped, CUDA): {e}");
// and the unbroken texts pass under the same calls
scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 1).unwrap();
scratch_text_check(&sample_cl, Dialect::OpenCl, 4, 64, 1).unwrap();
scratch_text_check(&sample_cu, Dialect::Cuda, 4, 64, 1).unwrap();
// a wrong slot count, RMW count or kernel count against a right text fails too (the check is tied to the class)
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 256, 1).is_err());
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 3, 64, 1).is_err());
assert!(scratch_text_check(&sample_metal, Dialect::Metal, 4, 64, 2).is_err());
}
// ---------------------------------------------------------------------------------------------------------------
// 5. The fuzz: 200 generated scratch programs, contract on every instruction, 4 units each across the 32-bit
// range including the wrap; with IGNEUM_SCRATCH_PACKS_OUT the packs for the Metal runs (question 3, 4)
// ---------------------------------------------------------------------------------------------------------------
/// Write a pack whose vectors.json carries `bases` (any number of units) instead of the three standard bases.
fn write_pack_with_bases(dir: &PathBuf, e: &Epoch, day: &str, bases: &[u32], source: &str) -> Vec<[u64; 32]> {
let mut pack = export_pack(e, day, source);
let outs: Vec<[u64; 32]> = bases.iter().map(|&b| e.hash_warp(b)).collect();
let vj = vectors_json(&e.program, day, e.dataset.log2_words, bases, &outs, &pack.vectors, e.dataset.mask, source, true);
for f in pack.files.iter_mut() {
if f.0 == "vectors.json" {
f.1 = vj.clone();
}
}
pack.write_to(dir).unwrap();
outs
}
fn contract(p: &Program) {
assert_eq!(p.instrs.len(), INSTR_COUNT);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Load).count() + p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), 16);
assert_eq!(p.instrs.iter().filter(|i| i.op == Op::Scratch).count(), p.class.scratch_slots());
assert!(p.instrs[0].op != Op::Load && p.instrs[0].op != Op::Scratch, "instruction 0 is never a memory op");
for (k, i) in p.instrs.iter().enumerate() {
assert!(i.src != i.dst, "#{k}: src == dst");
assert!((1..=31).contains(&i.rot), "#{k}: rot {}", i.rot);
assert!([1u8, 2, 4, 8, 16].contains(&i.mask), "#{k}: mask {}", i.mask);
assert!(i.dst < 8 && i.src < 8 && i.src2 < 8);
assert_eq!(i.width, 1, "#{k}: a scratch class reads one-word loads");
}
assert!(igneum_pow::accept::check(p).is_ok(), "an accepted program");
}
#[test]
fn fuzz_scr_programs_cpu() {
let n: usize = std::env::var("IGNEUM_SCRATCH_FUZZ").ok().and_then(|s| s.parse().ok()).unwrap_or(200);
// IGNEUM_FUZZ_SEED_BASE (default 0): the seed index offset for the continuous fuzzer (infra/build-server/capacity)
let base: usize = std::env::var("IGNEUM_FUZZ_SEED_BASE").ok().and_then(|s| s.parse().ok()).unwrap_or(0);
let out = std::env::var("IGNEUM_SCRATCH_PACKS_OUT").ok().map(PathBuf::from);
let mut rng = SplitMix64::new(0x6967_6e65_756d_2d73); // "igneum-s"
let day = "2026-10-03";
let closed = DatasetSource::new(day, DatasetMode::ClosedForm, 28);
// memory-hard sources per size, built once each (the cache fill is 0.2 s); only when packs are written
let mut mh: HashMap<u32, DatasetSource> = HashMap::new();
let mut manifest = String::from("pack\tclass\tlog2\tprogram_id\tscratch_ops_per_hash\tbases\n");
let mut per_class: HashMap<String, usize> = HashMap::new();
let mut units = 0usize;
let mut wraps = 0usize;
if let Some(dir) = &out {
std::fs::create_dir_all(dir).unwrap();
// the edge packs first: 64 MiB datasets (no dataset load in them), the four edge bases
for kb in [32u8, 128] {
for (name, _what, p) in edge_programs(kb) {
let log2 = 24;
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2));
let e = Epoch { program: p, dataset: ds };
let pack_name = format!("edge-{name}-k{kb}");
write_pack_with_bases(&dir.join(&pack_name), &e, day, &EDGE_BASES, "igneum-pow tests/scratch.rs edge");
manifest.push_str(&format!(
"{pack_name}\t{}\t{log2}\t{:016x}\t{}\t{}\n",
e.program.class.name(),
e.program.program_id(),
e.program.scratch_ops_per_hash(),
EDGE_BASES.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
));
mh.insert(log2, e.dataset);
}
}
}
for i in base..base + n {
let name = CLASSES[rng.below(CLASSES.len() as u64) as usize];
let c = class(name);
let seed = format!("igneum-scratch-fuzz/{i}");
let p = generate_class(&seed, c);
contract(&p);
*per_class.entry(name.to_string()).or_insert(0) += 1;
// four bases: one inside a 256-nonce batch (in-batch check on the GPU), one straddling 2^31, one in
// the last 256 nonces (the unit wraps past 2^32 or ends on it), one uniform
let b0 = (rng.below(8) as u32) * 32;
let b1 = 0x8000_0000u32.wrapping_sub(256).wrapping_add((rng.below(16) as u32) * 32);
let b2 = 0xffff_ff00u32.wrapping_add((rng.below(8) as u32) * 32);
let b3 = (rng.next() as u32) & !31;
let bases = [b0, b1, b2, b3];
// an aligned unit never straddles 2^32 (spec 1.9); the top unit ends on 0xffffffff and the persistent
// kernel's unit sequence wraps inside a launch, which the Metal run checks with packbench --batch-base
wraps += bases.iter().filter(|&&b| b >= 0xffff_ff00).count();
// the CPU: the interpreter is deterministic and every scratch event is inside the lane's slots
for &b in &bases {
let (r1, ev) = interpret_warp_scratch(&p, &p.seed, b, &closed, true);
let r2 = interpret_warp_scratch(&p, &p.seed, b, &closed, false).0;
assert_eq!(r1.hashes, r2.hashes);
assert_eq!(ev.len(), p.scratch_ops_per_hash() * LANES);
assert!(ev.iter().all(|e: &ScratchEvent| e.slot < c.scratch_slots_per_lane() as u32));
units += 1;
}
if let Some(dir) = &out {
let log2 = [24u32, 26, 28][rng.below(3) as usize];
let ds = mh.remove(&log2).unwrap_or_else(|| DatasetSource::new(day, DatasetMode::MemoryHard, log2));
let e = Epoch { program: p, dataset: ds };
let pack_name = format!("fuzz-{i:03}-{name}-l{log2}");
write_pack_with_bases(&dir.join(&pack_name), &e, day, &bases, "igneum-pow tests/scratch.rs fuzz");
manifest.push_str(&format!(
"{pack_name}\t{name}\t{log2}\t{:016x}\t{}\t{}\n",
e.program.program_id(),
e.program.scratch_ops_per_hash(),
bases.iter().map(|b| format!("{b}")).collect::<Vec<_>>().join(",")
));
mh.insert(log2, e.dataset);
} else {
let _ = rng.below(3);
}
}
let mut classes: Vec<_> = per_class.iter().collect();
classes.sort();
println!("fuzz: {n} programs, {units} units on the CPU, {wraps} units in the top 256 nonces, classes {classes:?}, seeds {base}..{}", base + n);
assert_eq!(units, 4 * n);
assert_eq!(wraps, n, "every program has a unit in the top 256 nonces");
if let Some(dir) = &out {
std::fs::write(dir.join("manifest.tsv"), manifest).unwrap();
println!("packs written to {}", dir.display());
}
}
/// The fold and rewrite, restated: a slot after `d` dependent RMWs holds 96 bits that are a function of the fill
/// (3 words, a pure function of nonce, slot and seed) and the `d` fold values; a chip that keeps the `d` fold
/// values (32 bits each) instead of the 96-bit slot recomputes the slot in `d` rewrites. This test pins the
/// arithmetic the analysis uses (question 2): the replay from the fold values reproduces the slot.
#[test]
fn slot_is_replayable_from_its_fold_values() {
let seed = seed_words_from_bytes(b"igneum-genesis");
let (base, lane, slot) = (0x1234_5600u32, 5u32, 17u32);
let fill = [scratch_fill(&seed, base, lane, slot, 0), scratch_fill(&seed, base, lane, slot, 1), scratch_fill(&seed, base, lane, slot, 2)];
let mut rng = SplitMix64::new(99);
let dsts: Vec<u32> = (0..64).map(|_| rng.next() as u32).collect();
// the honest sequence: read, fold, rewrite, 64 times
let mut w = fill;
let mut xs = Vec::new();
for &d in &dsts {
let x = fold_words(d, &w);
xs.push(x);
w = scratch_rewrite(x, &w);
}
// the replay: from the fill and the stored fold values alone
let mut w2 = fill;
for &x in &xs {
w2 = scratch_rewrite(x, &w2);
}
assert_eq!(w, w2);
// and nothing shorter: the fold value at step d depends on the slot content at step d, which depends on
// every earlier fold value (drop one and the chain diverges)
let mut w3 = fill;
for (i, &x) in xs.iter().enumerate() {
if i != 10 {
w3 = scratch_rewrite(x, &w3);
}
}
assert_ne!(w, w3);
}

View file

@ -0,0 +1,416 @@
{
"format": "igneum-program-pack-3",
"generator": 5,
"attempt": 0,
"program_id": "0xe5a4ac5978462156",
"program_id_derivation": "FNV-1a 64 over 'igneum-program/' || generator_le32 || seed_words as little-endian bytes || attempt_le32",
"dataset_mode": "memory-hard",
"seed": "igneum-epoch/4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925/day/69676e65756d2d6461792ffd50000000000000",
"seed_bytes": "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925",
"seed_words": ["0xbe8c5a0c", "0x7646e626", "0x508e29d0", "0xfb8a5b4a", "0x53c50955", "0x685d62fc", "0x6065c013", "0x881096eb"],
"seed_derivation": "seed_words = FNV-1a 64 over seed_bytes (attempt 0) or seed_bytes || attempt_le32 (attempt k >= 1), basis ^ (salt * 0x9E3779B97F4A7C15) for salt 0..3, then h ^= h>>33; h *= 0xff51afd7ed558ccd; h ^= h>>33; words[2*salt] = low 32, words[2*salt+1] = high 32",
"generator_rule": "version 2: exactly 16 load slots drawn first from instructions 1..63 (partial Fisher-Yates), the other 48 ops from the ten non-load weights (sum 75); a load's source is drawn from the registers other than dst written by an earlier instruction and not read by a load since; the candidate must pass the acceptance rule of spec 01 section 1.4.6 (static: no cyclically stale load source, every register has an injecting write; dynamic: 64 units on the seed-keyed closed-form dataset with no constant register bit, no lane-constant load site, under 164 saturated final values, every output bit within 136 of 1024, distinct addresses above 245760), else the next attempt of the seed is tried",
"lanes": 32,
"registers": 8,
"iterations": 8,
"instruction_count": 64,
"loads_per_hash": 128,
"program_class": "v5",
"era_seed_bytes": "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925",
"state": {
"block": "4020cb4382e3fe4b281c817c02582e147d8f851f566ae9172b28912b8e68b925",
"block_number": 0,
"root": "7e37a9fb19b154d32daf5bf30a50d339a75029fbc9eec9ea20e95439dba5a311",
"leaves": 11,
"records": 11,
"sampled": false,
"leaves_fnv1a64": "0xf141bfee2a8b11b0",
"leaf_derivation": "leaves[i] = Blake2b-512('igneum-sd1/' || root || i_le32 || record_i) as 16 little-endian words; item t XORs leaves[t mod leaves] into its 16 initial words before the first mixer",
"file": "leaves.bin"
},
"load_class": "mx8-eraaf3a9139+sh256x27+state",
"mixer_mult": 8,
"cache_growth": true,
"mixer": "class v3 (Counter ASIC 2.0, 5 October 2026, docs/plans/mixer-x4.md): every mixer application of the item derivation is 8 applications with round keys (r * 8 + j + 1) * 0x9E3779B9, the 8 dependent cache reads per item unchanged; cache growth rule option C: cache words = 2^(26 + doublings(day)), dataset words = 2^(genesis_log2 + doublings(day)), doublings(day) = floor(log2(1 + day / 1460)) for day = days since genesis",
"load_slots": 16,
"load_mix_percent_4_16_64": [100, 0, 0],
"load_width_counts_4_16_64": [16, 0, 0],
"bytes_per_hash": 512,
"wide_load": "read-width experiment (5 October 2026, docs/plans/read-width.md), NOT the lottery hash: a load of W words (width field, 4 or 16) reads dataset[b .. b + W) with b = (src & mask) & ~(W - 1) and folds every word into dst: x = dst ^ w[0]; for j in 1..W: x = (rotl(x, 11) * 0x9e3779b1) ^ w[j]; dst = x; width 1 is the plain load; the width is drawn per instruction from the class mix with one extra below(100) draw after the nine of version 2, and the program id is FNV-1a 64 over 'igneum-program-rw/' || generator_le32 || seed words || attempt_le32 || mix[3] || load_slots",
"era": {
"label": "af3a9139",
"seed_words": ["0xaf3a9139", "0x894db311", "0xd6173573", "0xc346bc0f", "0x2d547cec", "0xd5634ab8", "0xb9584827", "0x305d030e"],
"draw": "docs/plans/era-layout.md 1.1: SplitMix64 seeded with seed_words[0] | seed_words[1] << 32 of seed_words_from_bytes('igneum-era/' || n_le64 || E_n); width = allowed[below(|allowed|)], stride_mul = low32(next()) | 1, stride_rot = 1 + below(31), then four next() draws for a partial Fisher-Yates over positions log2(W)..15 of which 4 - log2(W) are used",
"allowed_widths": [1],
"width_words": 1,
"stride_mul": "0x2cb18b35",
"stride_rot": 9,
"interleave": [1, 9, 13, 14],
"address": "y = rotl(src * stride_mul, stride_rot); k = min(win, D - 26); idx = ((y & (mask >> k)) | ((off & (2^k - 1)) << (D - k))) & mask; a wide load aligns idx down to W words",
"windows": "per instruction, after the width roll: win = below(3), off = low32(next()) & (2^win - 1); used on a load slot (the instruction's win and off fields)",
"dataset_word": "dataset[w] = item(t(w))[j(w)]: j(w) gathers the bits of w at the interleave positions, t(w) is w with those bits removed",
"program_id_suffix": "'era/' || allowed[3] || width_words || stride_mul_le32 || stride_rot_le32 || interleave[4]"
},
"op_mix": {"load": 16, "add": 11, "shfl": 7, "mad": 6, "xor": 6, "mul": 4, "mulhi": 4, "rotr": 4, "rotl": 3, "sub": 2, "or": 1},
"register_init": "for i in 0..7: x = nonce ^ seed_words[i]; x += 0x9e3779b9 * (i+1) (mod 2^32); x = splitmix32(x); r[i] = x ^ seed_words[(i+1) & 7]",
"splitmix32": "x ^= x>>16; x *= 0x7feb352d; x ^= x>>15; x *= 0x846ca68b; x ^= x>>16",
"iteration": "sel = r0 sampled once at the top of each iteration, then all instructions in order",
"output": "lo = r0 ^ rotl(r1,7) ^ rotl(r2,14) ^ rotl(r3,21); hi = r4 ^ rotl(r5,9) ^ rotl(r6,18) ^ rotl(r7,27); out = (hi << 32) | lo",
"op_semantics": {
"add": "dst = dst + src + (bit `bit` of sel ? imm2 : imm)",
"sub": "dst = dst - src",
"mul": "dst = dst * src (low 32)",
"mulhi": "dst = high 32 bits of dst * src",
"xor": "dst = dst ^ src",
"or": "dst = dst | src",
"rotl": "dst = rotl(dst, rot), rot in 1..31",
"rotr": "dst = rotr(dst, src & 31)",
"mad": "dst = src * src2 + dst",
"shfl": "dst = dst ^ (src of lane (lane ^ mask)), mask in {1,2,4,8,16}, within the 32-lane warp",
"load": "dst = dst ^ dataset[src & dataset.mask]",
"wload": "base = (src of lane 0 & dataset.mask) & ~31; dst = dst ^ dataset[base + lane] (warp-coalesced 128-byte load, lever b, only when --wide-frac > 0)"
},
"dataset": {
"log2_words": 28,
"bytes": 1073741824,
"mask": "0x0fffffff",
"day": "bytes:69676e65756d2d6461792ffd50000000000000",
"day_bytes": "69676e65756d2d6461792ffd50000000000000",
"day_words_from": "seed_words_from_bytes(day_bytes)",
"d0": "0xe23f9008",
"d1": "0xde33e763",
"mode": "memory-hard",
"spec": "proto-metal/MEMHARD.md",
"key": ["0xe23f9008", "0xde33e763", "0xc5ba415d", "0x8ddf6786", "0x59f4a4be", "0x8a3bc680", "0x701b8e40", "0x3b59025a"],
"key_derivation": "the 8 words of seed_words_from_bytes(day_bytes); d0, d1 are key[0], key[1]",
"cache": {"log2_words": 26, "bytes": 268435456, "line_words": 16, "segment_lines": 64, "segments": 65536, "block": "ChaCha12 core + feed-forward, rotations 16 12 8 7", "sigma": ["0x61707865", "0x3320646e", "0x79622d32", "0x6b206574"], "tag": ["0x49676e65", "0x756d4d48"], "chain": "in_j = prev_line ^ (sigma[0..3] || key[0..7] || seg || j || tag[0..1]); line_j = block(in_j); prev_0 = 0"},
"mixer": {"draw": "SplitMix64 seeded with key[0] | key[1] << 32: rot[0..7] = 1 + next() % 31, mul[0..15] = low32(next()) | 1, rc[0..15] = low32(next())", "rot": [28, 11, 18, 18, 13, 28, 5, 13], "mul": ["0xd1d79e7f", "0xbcb61f8b", "0x970e91ab", "0xd749d96d", "0x143d7339", "0x2cde0d69", "0x12d0a8c1", "0x2d335be9", "0x80a8aae9", "0x7ac896c7", "0x9de23db7", "0xc362827d", "0x5f4cdb5b", "0xfc6c5097", "0x6f547f83", "0x1ac31b47"], "rc": ["0xe5bef5a3", "0xdb2a7d90", "0xcd1fa7e1", "0x30289419", "0x0a730d58", "0x432f8579", "0x6ab978a5", "0x8c984f49", "0x788c3c9e", "0x051eef02", "0x05e77db9", "0x8b3cd4f3", "0x6948d6cf", "0x0d8677f6", "0xf9505c8c", "0x4513c163"], "round": "for i in 0..15: s[i] = (s[i] ^ (rc[i] + (r+1) * 0x9E3779B9)) * mul[i]; then quarter rounds on columns (0,4,8,12) (1,5,9,13) (2,6,10,14) (3,7,11,15) with rot[0..3] and diagonals (0,5,10,15) (1,6,11,12) (2,7,8,13) (3,4,9,14) with rot[4..7]", "quarter_round": "a += b; d ^= a; d = rotl(d, r1); c += d; b ^= c; b = rotl(b, r2); a += b; d ^= a; d = rotl(d, r3); c += d; b ^= c; b = rotl(b, r4)"},
"mixer_mult": 8,
"item": "s[0..7] = key; s[8+i] = t * mul[i] + rc[i] for i in 0..7; for r in 0..7: for j in 0..7: s = M(s, rk = (r * 8 + j + 1) * 0x9E3779B9); line = s[0] & 0x003fffff; s[i] ^= cache[line * 16 + i]; then for j in 0..7: s = M(s, rk = (64 + j + 1) * 0x9E3779B9); item(t) = s",
"word": "dataset[w] = item(w >> 4)[w & 15]"
},
"shadow": {"instrs": 256, "reps": 27, "instrs_per_hash": 55296, "op_mix": {"add": 45, "mad": 31, "shfl": 31, "xor": 30, "rotl": 28, "mul": 22, "or": 18, "mulhi": 17, "rotr": 17, "sub": 17}, "rule": "Counter ASIC 3.0 item 8 (docs/analysis/latency-shadow-2026-10-06.md): after the 64 base instructions the program stream draws instrs more ALU instructions (the non-load table, nine draws each, the source as on an ALU slot); the block runs reps times at the end of every iteration with the iteration's sel; the acceptance rule interprets the base program only", "program_id_suffix": "'shadow/' || instrs_le16 || reps_le16", "instructions": [
{"i": 0, "op": "or", "dst": 6, "src": 0, "src2": 2, "imm": "0xc1eb1523", "imm2": "0x925958ea", "rot": 14, "bit": 14, "mask": 8},
{"i": 1, "op": "mul", "dst": 7, "src": 3, "src2": 4, "imm": "0xc7044334", "imm2": "0x755566b6", "rot": 19, "bit": 17, "mask": 8},
{"i": 2, "op": "mad", "dst": 3, "src": 5, "src2": 5, "imm": "0x33b52f17", "imm2": "0xcc459a7a", "rot": 5, "bit": 19, "mask": 16},
{"i": 3, "op": "add", "dst": 7, "src": 4, "src2": 4, "imm": "0x97d3d105", "imm2": "0x38e58a06", "rot": 11, "bit": 6, "mask": 1},
{"i": 4, "op": "add", "dst": 6, "src": 0, "src2": 6, "imm": "0x29b87b9a", "imm2": "0x9629673d", "rot": 16, "bit": 20, "mask": 8},
{"i": 5, "op": "shfl", "dst": 6, "src": 2, "src2": 5, "imm": "0xb61f2e77", "imm2": "0x30e3807c", "rot": 1, "bit": 7, "mask": 1},
{"i": 6, "op": "mad", "dst": 5, "src": 3, "src2": 5, "imm": "0x0de66253", "imm2": "0xfdfdd03e", "rot": 4, "bit": 27, "mask": 4},
{"i": 7, "op": "rotl", "dst": 3, "src": 7, "src2": 1, "imm": "0x688514dd", "imm2": "0x139d08b5", "rot": 5, "bit": 1, "mask": 1},
{"i": 8, "op": "add", "dst": 0, "src": 7, "src2": 3, "imm": "0xa7c1fb0f", "imm2": "0x4a5f55e1", "rot": 15, "bit": 12, "mask": 1},
{"i": 9, "op": "mul", "dst": 5, "src": 7, "src2": 2, "imm": "0x1a64de5a", "imm2": "0xe37b0875", "rot": 5, "bit": 27, "mask": 16},
{"i": 10, "op": "xor", "dst": 4, "src": 5, "src2": 3, "imm": "0x750d4f1a", "imm2": "0x2f3d9402", "rot": 3, "bit": 14, "mask": 2},
{"i": 11, "op": "mulhi", "dst": 6, "src": 4, "src2": 5, "imm": "0x2eaa86c6", "imm2": "0x9fd5b22f", "rot": 30, "bit": 10, "mask": 1},
{"i": 12, "op": "rotl", "dst": 5, "src": 2, "src2": 4, "imm": "0x8df50473", "imm2": "0x23d780ce", "rot": 13, "bit": 9, "mask": 1},
{"i": 13, "op": "mad", "dst": 0, "src": 2, "src2": 1, "imm": "0xa956a7e6", "imm2": "0x58dbb356", "rot": 31, "bit": 11, "mask": 8},
{"i": 14, "op": "mulhi", "dst": 2, "src": 4, "src2": 6, "imm": "0x6fbfb145", "imm2": "0x064eb015", "rot": 20, "bit": 2, "mask": 4},
{"i": 15, "op": "or", "dst": 3, "src": 1, "src2": 5, "imm": "0xd313bc27", "imm2": "0x9f0fc235", "rot": 17, "bit": 16, "mask": 1},
{"i": 16, "op": "sub", "dst": 0, "src": 5, "src2": 0, "imm": "0x86cbd336", "imm2": "0x94715167", "rot": 17, "bit": 31, "mask": 8},
{"i": 17, "op": "mul", "dst": 0, "src": 6, "src2": 3, "imm": "0x95911c6b", "imm2": "0x816cea32", "rot": 23, "bit": 1, "mask": 2},
{"i": 18, "op": "shfl", "dst": 5, "src": 1, "src2": 3, "imm": "0xf563dcad", "imm2": "0xb2889856", "rot": 14, "bit": 24, "mask": 4},
{"i": 19, "op": "or", "dst": 3, "src": 4, "src2": 2, "imm": "0x7bb48d19", "imm2": "0xbf95ad6f", "rot": 8, "bit": 30, "mask": 16},
{"i": 20, "op": "mad", "dst": 6, "src": 3, "src2": 2, "imm": "0xeb686a8d", "imm2": "0x8bda9f8a", "rot": 11, "bit": 18, "mask": 1},
{"i": 21, "op": "rotl", "dst": 0, "src": 6, "src2": 0, "imm": "0x0915df51", "imm2": "0xa1365cda", "rot": 11, "bit": 25, "mask": 1},
{"i": 22, "op": "rotl", "dst": 3, "src": 0, "src2": 3, "imm": "0xc09e4cdc", "imm2": "0x50c1369d", "rot": 25, "bit": 31, "mask": 2},
{"i": 23, "op": "xor", "dst": 6, "src": 4, "src2": 1, "imm": "0xce01abec", "imm2": "0x5ff8c52e", "rot": 30, "bit": 19, "mask": 8},
{"i": 24, "op": "or", "dst": 4, "src": 2, "src2": 4, "imm": "0xdacf9e4b", "imm2": "0x086a5684", "rot": 2, "bit": 10, "mask": 2},
{"i": 25, "op": "add", "dst": 1, "src": 6, "src2": 3, "imm": "0x53e74799", "imm2": "0xfc2f6d17", "rot": 8, "bit": 7, "mask": 16},
{"i": 26, "op": "rotr", "dst": 7, "src": 6, "src2": 7, "imm": "0xc89a75a0", "imm2": "0x67813253", "rot": 24, "bit": 13, "mask": 1},
{"i": 27, "op": "add", "dst": 7, "src": 5, "src2": 4, "imm": "0xb8fae90e", "imm2": "0x787048dc", "rot": 26, "bit": 27, "mask": 1},
{"i": 28, "op": "xor", "dst": 4, "src": 1, "src2": 0, "imm": "0x4c90c2ea", "imm2": "0x2ce7865b", "rot": 4, "bit": 7, "mask": 2},
{"i": 29, "op": "xor", "dst": 6, "src": 1, "src2": 6, "imm": "0x6131ab08", "imm2": "0xd1825676", "rot": 4, "bit": 30, "mask": 2},
{"i": 30, "op": "mulhi", "dst": 2, "src": 0, "src2": 1, "imm": "0xbe4a545f", "imm2": "0xdc29e823", "rot": 17, "bit": 5, "mask": 8},
{"i": 31, "op": "rotr", "dst": 1, "src": 4, "src2": 1, "imm": "0x4d2f9cdc", "imm2": "0xdee1cc73", "rot": 5, "bit": 16, "mask": 4},
{"i": 32, "op": "xor", "dst": 7, "src": 4, "src2": 5, "imm": "0x9d277aaa", "imm2": "0xa902fca5", "rot": 11, "bit": 3, "mask": 1},
{"i": 33, "op": "mad", "dst": 3, "src": 2, "src2": 6, "imm": "0x7db09644", "imm2": "0x077bf3d9", "rot": 30, "bit": 19, "mask": 2},
{"i": 34, "op": "add", "dst": 3, "src": 1, "src2": 1, "imm": "0xd2dc42eb", "imm2": "0x9ffd510b", "rot": 17, "bit": 12, "mask": 2},
{"i": 35, "op": "add", "dst": 4, "src": 7, "src2": 0, "imm": "0xad8eda6f", "imm2": "0xf871ba37", "rot": 24, "bit": 7, "mask": 4},
{"i": 36, "op": "add", "dst": 7, "src": 4, "src2": 3, "imm": "0xb0cef8f4", "imm2": "0xd02d30ca", "rot": 22, "bit": 3, "mask": 16},
{"i": 37, "op": "mad", "dst": 4, "src": 3, "src2": 2, "imm": "0x6ed99a66", "imm2": "0x83dc97c0", "rot": 4, "bit": 1, "mask": 4},
{"i": 38, "op": "mad", "dst": 5, "src": 7, "src2": 7, "imm": "0xe24fdd24", "imm2": "0x360f5c8b", "rot": 19, "bit": 9, "mask": 8},
{"i": 39, "op": "add", "dst": 3, "src": 6, "src2": 4, "imm": "0x2bd506e2", "imm2": "0x82e98a22", "rot": 28, "bit": 26, "mask": 1},
{"i": 40, "op": "rotl", "dst": 0, "src": 2, "src2": 1, "imm": "0xdb1faed3", "imm2": "0xf968a20d", "rot": 21, "bit": 30, "mask": 16},
{"i": 41, "op": "or", "dst": 1, "src": 0, "src2": 1, "imm": "0xe6dfe522", "imm2": "0xb75813da", "rot": 10, "bit": 13, "mask": 1},
{"i": 42, "op": "add", "dst": 7, "src": 5, "src2": 7, "imm": "0xc0386e49", "imm2": "0x6435f6d8", "rot": 26, "bit": 28, "mask": 16},
{"i": 43, "op": "mulhi", "dst": 1, "src": 0, "src2": 4, "imm": "0xda497f7d", "imm2": "0xc4b748f4", "rot": 30, "bit": 10, "mask": 4},
{"i": 44, "op": "add", "dst": 5, "src": 6, "src2": 0, "imm": "0x3a8921a4", "imm2": "0x1f682c68", "rot": 12, "bit": 12, "mask": 1},
{"i": 45, "op": "add", "dst": 4, "src": 5, "src2": 6, "imm": "0xeeca4334", "imm2": "0xf246e180", "rot": 26, "bit": 1, "mask": 1},
{"i": 46, "op": "xor", "dst": 2, "src": 3, "src2": 5, "imm": "0x052178c3", "imm2": "0xaab31f10", "rot": 28, "bit": 16, "mask": 8},
{"i": 47, "op": "xor", "dst": 2, "src": 5, "src2": 0, "imm": "0x91fd7273", "imm2": "0xacc42b07", "rot": 22, "bit": 27, "mask": 4},
{"i": 48, "op": "mad", "dst": 7, "src": 4, "src2": 1, "imm": "0x05a2970f", "imm2": "0x14876824", "rot": 10, "bit": 5, "mask": 1},
{"i": 49, "op": "mad", "dst": 2, "src": 6, "src2": 1, "imm": "0x4440d6de", "imm2": "0x352023f9", "rot": 10, "bit": 25, "mask": 8},
{"i": 50, "op": "mad", "dst": 7, "src": 0, "src2": 7, "imm": "0xa6f82633", "imm2": "0x78465772", "rot": 13, "bit": 8, "mask": 4},
{"i": 51, "op": "add", "dst": 5, "src": 6, "src2": 5, "imm": "0x745776d8", "imm2": "0x41bd68cc", "rot": 29, "bit": 26, "mask": 1},
{"i": 52, "op": "shfl", "dst": 3, "src": 0, "src2": 6, "imm": "0x8f94ebac", "imm2": "0xe4a208a0", "rot": 5, "bit": 16, "mask": 1},
{"i": 53, "op": "or", "dst": 4, "src": 0, "src2": 5, "imm": "0x7b1235f3", "imm2": "0x408b4ea9", "rot": 27, "bit": 19, "mask": 8},
{"i": 54, "op": "mad", "dst": 0, "src": 2, "src2": 4, "imm": "0x30114050", "imm2": "0xe71dff8f", "rot": 6, "bit": 30, "mask": 1},
{"i": 55, "op": "rotl", "dst": 5, "src": 1, "src2": 7, "imm": "0x69f92198", "imm2": "0x65496f29", "rot": 9, "bit": 20, "mask": 4},
{"i": 56, "op": "mul", "dst": 5, "src": 7, "src2": 6, "imm": "0x88b884d4", "imm2": "0x041b67f7", "rot": 25, "bit": 8, "mask": 16},
{"i": 57, "op": "rotl", "dst": 7, "src": 5, "src2": 6, "imm": "0x7d1e18bc", "imm2": "0xf54ade27", "rot": 27, "bit": 20, "mask": 1},
{"i": 58, "op": "sub", "dst": 7, "src": 6, "src2": 4, "imm": "0xa457bbab", "imm2": "0x4b05eb29", "rot": 6, "bit": 6, "mask": 4},
{"i": 59, "op": "rotl", "dst": 2, "src": 7, "src2": 7, "imm": "0x4890bdda", "imm2": "0xadcb3d99", "rot": 22, "bit": 26, "mask": 16},
{"i": 60, "op": "shfl", "dst": 7, "src": 1, "src2": 7, "imm": "0xe4da58eb", "imm2": "0x1d50080d", "rot": 22, "bit": 26, "mask": 2},
{"i": 61, "op": "add", "dst": 4, "src": 0, "src2": 5, "imm": "0x60d0ff55", "imm2": "0xcba6a161", "rot": 6, "bit": 7, "mask": 8},
{"i": 62, "op": "mul", "dst": 1, "src": 5, "src2": 1, "imm": "0x0fdc702e", "imm2": "0x6b1bd03a", "rot": 9, "bit": 23, "mask": 4},
{"i": 63, "op": "mulhi", "dst": 7, "src": 6, "src2": 1, "imm": "0xcc44ed62", "imm2": "0x931dd006", "rot": 1, "bit": 0, "mask": 2},
{"i": 64, "op": "mul", "dst": 5, "src": 0, "src2": 7, "imm": "0x448c3aba", "imm2": "0x1136818a", "rot": 23, "bit": 7, "mask": 1},
{"i": 65, "op": "rotr", "dst": 2, "src": 5, "src2": 1, "imm": "0xb520dfb8", "imm2": "0xdce7d8f4", "rot": 22, "bit": 10, "mask": 8},
{"i": 66, "op": "mulhi", "dst": 0, "src": 3, "src2": 4, "imm": "0x19fada48", "imm2": "0x3cc04e0e", "rot": 1, "bit": 24, "mask": 16},
{"i": 67, "op": "or", "dst": 0, "src": 7, "src2": 3, "imm": "0x1c5905b4", "imm2": "0xf896dfb7", "rot": 31, "bit": 30, "mask": 8},
{"i": 68, "op": "rotl", "dst": 0, "src": 1, "src2": 5, "imm": "0xa28511c1", "imm2": "0xbf649695", "rot": 19, "bit": 27, "mask": 1},
{"i": 69, "op": "rotl", "dst": 0, "src": 7, "src2": 5, "imm": "0x22e2fef3", "imm2": "0x468bec9e", "rot": 1, "bit": 28, "mask": 1},
{"i": 70, "op": "or", "dst": 2, "src": 3, "src2": 6, "imm": "0xcb542821", "imm2": "0xe084cf11", "rot": 15, "bit": 5, "mask": 8},
{"i": 71, "op": "mad", "dst": 3, "src": 7, "src2": 5, "imm": "0xf06b8f6a", "imm2": "0x7382e15f", "rot": 6, "bit": 28, "mask": 16},
{"i": 72, "op": "mad", "dst": 7, "src": 0, "src2": 4, "imm": "0x4f64c588", "imm2": "0x47b63ee9", "rot": 8, "bit": 10, "mask": 2},
{"i": 73, "op": "or", "dst": 4, "src": 3, "src2": 2, "imm": "0xff138eea", "imm2": "0x6fd90eda", "rot": 23, "bit": 3, "mask": 16},
{"i": 74, "op": "xor", "dst": 3, "src": 1, "src2": 0, "imm": "0x6011ece1", "imm2": "0x89579f15", "rot": 11, "bit": 2, "mask": 4},
{"i": 75, "op": "add", "dst": 1, "src": 3, "src2": 4, "imm": "0x3f4ffea2", "imm2": "0x8e35924e", "rot": 12, "bit": 15, "mask": 2},
{"i": 76, "op": "mul", "dst": 1, "src": 3, "src2": 7, "imm": "0x15efa846", "imm2": "0xa1c971f1", "rot": 14, "bit": 20, "mask": 1},
{"i": 77, "op": "mad", "dst": 0, "src": 3, "src2": 2, "imm": "0x8660530a", "imm2": "0xa013524b", "rot": 14, "bit": 12, "mask": 8},
{"i": 78, "op": "shfl", "dst": 2, "src": 0, "src2": 0, "imm": "0xcc2e96bd", "imm2": "0xe32c6f87", "rot": 10, "bit": 11, "mask": 16},
{"i": 79, "op": "mulhi", "dst": 1, "src": 0, "src2": 6, "imm": "0x61931e3f", "imm2": "0x86383e7f", "rot": 17, "bit": 3, "mask": 1},
{"i": 80, "op": "mulhi", "dst": 0, "src": 4, "src2": 7, "imm": "0xd9991afc", "imm2": "0xd00e8678", "rot": 24, "bit": 15, "mask": 16},
{"i": 81, "op": "mul", "dst": 4, "src": 2, "src2": 3, "imm": "0x472f70f4", "imm2": "0x9fde4236", "rot": 10, "bit": 5, "mask": 4},
{"i": 82, "op": "sub", "dst": 5, "src": 7, "src2": 6, "imm": "0x55a7acb1", "imm2": "0x9fde8df1", "rot": 30, "bit": 29, "mask": 16},
{"i": 83, "op": "mad", "dst": 0, "src": 2, "src2": 3, "imm": "0x1c487c40", "imm2": "0xd9d9a1c3", "rot": 9, "bit": 5, "mask": 4},
{"i": 84, "op": "sub", "dst": 7, "src": 2, "src2": 4, "imm": "0x515c0a72", "imm2": "0x6df5c83d", "rot": 28, "bit": 21, "mask": 4},
{"i": 85, "op": "xor", "dst": 1, "src": 2, "src2": 4, "imm": "0x917f1174", "imm2": "0x8e72e519", "rot": 9, "bit": 18, "mask": 16},
{"i": 86, "op": "shfl", "dst": 6, "src": 3, "src2": 3, "imm": "0xda856cc2", "imm2": "0x932ae8d8", "rot": 28, "bit": 1, "mask": 16},
{"i": 87, "op": "mad", "dst": 3, "src": 0, "src2": 0, "imm": "0x4ef22321", "imm2": "0x5666ada3", "rot": 4, "bit": 4, "mask": 1},
{"i": 88, "op": "rotl", "dst": 0, "src": 1, "src2": 3, "imm": "0xbb3ec00c", "imm2": "0xffe78937", "rot": 20, "bit": 19, "mask": 4},
{"i": 89, "op": "sub", "dst": 1, "src": 5, "src2": 4, "imm": "0x8f295708", "imm2": "0x6608577f", "rot": 17, "bit": 18, "mask": 16},
{"i": 90, "op": "shfl", "dst": 6, "src": 4, "src2": 5, "imm": "0xd953aa15", "imm2": "0x4ef9f062", "rot": 19, "bit": 9, "mask": 16},
{"i": 91, "op": "xor", "dst": 5, "src": 1, "src2": 5, "imm": "0x6c02cde2", "imm2": "0x118dbd2b", "rot": 2, "bit": 20, "mask": 16},
{"i": 92, "op": "rotl", "dst": 5, "src": 0, "src2": 5, "imm": "0x9f0b0c61", "imm2": "0xe26dbe7a", "rot": 12, "bit": 17, "mask": 2},
{"i": 93, "op": "mad", "dst": 6, "src": 1, "src2": 2, "imm": "0x78a44a91", "imm2": "0x3ce118a9", "rot": 18, "bit": 16, "mask": 16},
{"i": 94, "op": "rotr", "dst": 2, "src": 5, "src2": 3, "imm": "0x02295fcb", "imm2": "0xc3353c64", "rot": 5, "bit": 31, "mask": 16},
{"i": 95, "op": "or", "dst": 0, "src": 4, "src2": 4, "imm": "0x727120c3", "imm2": "0x528f4416", "rot": 6, "bit": 20, "mask": 8},
{"i": 96, "op": "mul", "dst": 7, "src": 3, "src2": 4, "imm": "0x919c42ea", "imm2": "0x87526aa5", "rot": 28, "bit": 1, "mask": 8},
{"i": 97, "op": "or", "dst": 5, "src": 0, "src2": 7, "imm": "0x8a0a100c", "imm2": "0xb2c65cc3", "rot": 13, "bit": 7, "mask": 4},
{"i": 98, "op": "add", "dst": 3, "src": 1, "src2": 1, "imm": "0xbcfdd8a7", "imm2": "0x386bb642", "rot": 8, "bit": 13, "mask": 8},
{"i": 99, "op": "add", "dst": 0, "src": 6, "src2": 7, "imm": "0xc921a021", "imm2": "0xa3962d03", "rot": 31, "bit": 9, "mask": 1},
{"i": 100, "op": "rotr", "dst": 1, "src": 0, "src2": 3, "imm": "0x918c6f69", "imm2": "0x1a125801", "rot": 18, "bit": 29, "mask": 16},
{"i": 101, "op": "add", "dst": 2, "src": 4, "src2": 1, "imm": "0x70da4067", "imm2": "0xc5990c5c", "rot": 3, "bit": 18, "mask": 4},
{"i": 102, "op": "rotl", "dst": 2, "src": 7, "src2": 0, "imm": "0x9048d081", "imm2": "0xc25a9f56", "rot": 3, "bit": 25, "mask": 16},
{"i": 103, "op": "shfl", "dst": 1, "src": 4, "src2": 7, "imm": "0x70ca1427", "imm2": "0x6632093d", "rot": 18, "bit": 28, "mask": 4},
{"i": 104, "op": "xor", "dst": 0, "src": 2, "src2": 0, "imm": "0x0637204c", "imm2": "0x3b1b7f94", "rot": 1, "bit": 12, "mask": 2},
{"i": 105, "op": "add", "dst": 4, "src": 3, "src2": 5, "imm": "0x1a5880cb", "imm2": "0xb15aec75", "rot": 31, "bit": 28, "mask": 1},
{"i": 106, "op": "shfl", "dst": 4, "src": 0, "src2": 4, "imm": "0x7c22cb00", "imm2": "0xc4df7949", "rot": 3, "bit": 15, "mask": 2},
{"i": 107, "op": "shfl", "dst": 7, "src": 0, "src2": 0, "imm": "0xd43c76ca", "imm2": "0x8d5b9751", "rot": 31, "bit": 28, "mask": 4},
{"i": 108, "op": "mad", "dst": 1, "src": 7, "src2": 3, "imm": "0x69879993", "imm2": "0x3c4e18b0", "rot": 19, "bit": 25, "mask": 4},
{"i": 109, "op": "mad", "dst": 0, "src": 2, "src2": 3, "imm": "0x8ce3a4a6", "imm2": "0x1b3f0b74", "rot": 19, "bit": 8, "mask": 2},
{"i": 110, "op": "or", "dst": 2, "src": 4, "src2": 5, "imm": "0xcdbdac98", "imm2": "0x8d0e9f92", "rot": 19, "bit": 13, "mask": 1},
{"i": 111, "op": "shfl", "dst": 6, "src": 1, "src2": 1, "imm": "0xe74157f6", "imm2": "0xf92dead6", "rot": 4, "bit": 14, "mask": 1},
{"i": 112, "op": "mad", "dst": 2, "src": 5, "src2": 3, "imm": "0x200831bc", "imm2": "0x1ee98048", "rot": 29, "bit": 14, "mask": 1},
{"i": 113, "op": "mulhi", "dst": 4, "src": 0, "src2": 7, "imm": "0xd6827ead", "imm2": "0xc0ee1933", "rot": 26, "bit": 23, "mask": 16},
{"i": 114, "op": "rotl", "dst": 7, "src": 2, "src2": 2, "imm": "0x6ae1be2b", "imm2": "0x83216a4e", "rot": 29, "bit": 25, "mask": 16},
{"i": 115, "op": "sub", "dst": 3, "src": 7, "src2": 1, "imm": "0xa1d1646e", "imm2": "0x6fa998ce", "rot": 15, "bit": 29, "mask": 2},
{"i": 116, "op": "mulhi", "dst": 0, "src": 1, "src2": 4, "imm": "0xd7921c15", "imm2": "0x8b2345a7", "rot": 10, "bit": 19, "mask": 4},
{"i": 117, "op": "rotl", "dst": 4, "src": 0, "src2": 4, "imm": "0xc1b1b0ab", "imm2": "0xd19b3f39", "rot": 12, "bit": 16, "mask": 1},
{"i": 118, "op": "add", "dst": 1, "src": 4, "src2": 5, "imm": "0x24494ad3", "imm2": "0xc5e98ee5", "rot": 22, "bit": 7, "mask": 16},
{"i": 119, "op": "add", "dst": 0, "src": 4, "src2": 6, "imm": "0x1b2d1c83", "imm2": "0xa97bc949", "rot": 8, "bit": 4, "mask": 16},
{"i": 120, "op": "shfl", "dst": 5, "src": 4, "src2": 7, "imm": "0xb93015ff", "imm2": "0xc22366cd", "rot": 23, "bit": 16, "mask": 4},
{"i": 121, "op": "shfl", "dst": 3, "src": 2, "src2": 5, "imm": "0xc0aaacdb", "imm2": "0xaaa7d229", "rot": 31, "bit": 29, "mask": 8},
{"i": 122, "op": "rotl", "dst": 7, "src": 1, "src2": 3, "imm": "0xcce6a73f", "imm2": "0x517b6b86", "rot": 3, "bit": 16, "mask": 8},
{"i": 123, "op": "rotr", "dst": 4, "src": 6, "src2": 0, "imm": "0x8e156aca", "imm2": "0x0a894b3e", "rot": 2, "bit": 12, "mask": 2},
{"i": 124, "op": "shfl", "dst": 6, "src": 5, "src2": 2, "imm": "0xd88c8446", "imm2": "0xf49fedf3", "rot": 3, "bit": 23, "mask": 1},
{"i": 125, "op": "rotl", "dst": 5, "src": 2, "src2": 0, "imm": "0xd10b27cf", "imm2": "0x2dde1faa", "rot": 21, "bit": 7, "mask": 2},
{"i": 126, "op": "mad", "dst": 0, "src": 3, "src2": 1, "imm": "0x619e1818", "imm2": "0x62add310", "rot": 26, "bit": 1, "mask": 4},
{"i": 127, "op": "add", "dst": 0, "src": 3, "src2": 6, "imm": "0xff14c08d", "imm2": "0xa7f6bb54", "rot": 25, "bit": 24, "mask": 4},
{"i": 128, "op": "xor", "dst": 7, "src": 4, "src2": 4, "imm": "0x34224086", "imm2": "0x5e9f858e", "rot": 4, "bit": 16, "mask": 8},
{"i": 129, "op": "mulhi", "dst": 3, "src": 5, "src2": 6, "imm": "0x89c3cbdd", "imm2": "0x4e80334c", "rot": 29, "bit": 4, "mask": 1},
{"i": 130, "op": "shfl", "dst": 5, "src": 7, "src2": 1, "imm": "0x37d4c3e2", "imm2": "0xbd54378c", "rot": 14, "bit": 26, "mask": 8},
{"i": 131, "op": "mulhi", "dst": 6, "src": 4, "src2": 3, "imm": "0x4f5fac7c", "imm2": "0x74070028", "rot": 5, "bit": 25, "mask": 8},
{"i": 132, "op": "xor", "dst": 5, "src": 3, "src2": 0, "imm": "0x690eb003", "imm2": "0x8516a581", "rot": 20, "bit": 21, "mask": 2},
{"i": 133, "op": "rotl", "dst": 0, "src": 5, "src2": 7, "imm": "0x42b31885", "imm2": "0x08249acb", "rot": 19, "bit": 6, "mask": 8},
{"i": 134, "op": "rotr", "dst": 4, "src": 3, "src2": 3, "imm": "0x07d729ed", "imm2": "0xf564da8a", "rot": 17, "bit": 9, "mask": 16},
{"i": 135, "op": "shfl", "dst": 1, "src": 2, "src2": 0, "imm": "0x19130ef2", "imm2": "0xf59815e5", "rot": 17, "bit": 12, "mask": 16},
{"i": 136, "op": "sub", "dst": 5, "src": 2, "src2": 5, "imm": "0xdeb56168", "imm2": "0x871104c4", "rot": 13, "bit": 17, "mask": 16},
{"i": 137, "op": "xor", "dst": 3, "src": 5, "src2": 4, "imm": "0x884fc7da", "imm2": "0x83a4a5bf", "rot": 23, "bit": 12, "mask": 4},
{"i": 138, "op": "sub", "dst": 0, "src": 2, "src2": 6, "imm": "0x3e67a6fd", "imm2": "0x844d2039", "rot": 3, "bit": 24, "mask": 4},
{"i": 139, "op": "mul", "dst": 3, "src": 7, "src2": 7, "imm": "0x965933b4", "imm2": "0xf37ef93d", "rot": 6, "bit": 9, "mask": 4},
{"i": 140, "op": "add", "dst": 5, "src": 3, "src2": 3, "imm": "0x82ab98f4", "imm2": "0x2d465ca8", "rot": 11, "bit": 24, "mask": 1},
{"i": 141, "op": "add", "dst": 4, "src": 1, "src2": 7, "imm": "0x80594390", "imm2": "0x237a9d8e", "rot": 14, "bit": 11, "mask": 8},
{"i": 142, "op": "mad", "dst": 3, "src": 7, "src2": 0, "imm": "0xd37861a1", "imm2": "0x2fe70eae", "rot": 2, "bit": 11, "mask": 16},
{"i": 143, "op": "or", "dst": 0, "src": 3, "src2": 5, "imm": "0x1a018e2c", "imm2": "0x301aa5f6", "rot": 15, "bit": 26, "mask": 1},
{"i": 144, "op": "mulhi", "dst": 0, "src": 1, "src2": 5, "imm": "0x69d22dfc", "imm2": "0x4a90cb1e", "rot": 21, "bit": 9, "mask": 16},
{"i": 145, "op": "xor", "dst": 6, "src": 2, "src2": 0, "imm": "0xf3f41f47", "imm2": "0xdf8d33e9", "rot": 22, "bit": 0, "mask": 8},
{"i": 146, "op": "shfl", "dst": 7, "src": 1, "src2": 1, "imm": "0xf8a52377", "imm2": "0x0613ef41", "rot": 24, "bit": 12, "mask": 16},
{"i": 147, "op": "add", "dst": 0, "src": 1, "src2": 3, "imm": "0x07f15148", "imm2": "0xe68cf57e", "rot": 2, "bit": 18, "mask": 4},
{"i": 148, "op": "or", "dst": 7, "src": 3, "src2": 0, "imm": "0xbfd39b49", "imm2": "0xc7f6e97c", "rot": 13, "bit": 27, "mask": 2},
{"i": 149, "op": "xor", "dst": 4, "src": 2, "src2": 7, "imm": "0x47409e43", "imm2": "0x04ef7db9", "rot": 8, "bit": 6, "mask": 4},
{"i": 150, "op": "xor", "dst": 7, "src": 1, "src2": 1, "imm": "0x5d42a156", "imm2": "0xc2f77c48", "rot": 27, "bit": 17, "mask": 4},
{"i": 151, "op": "rotr", "dst": 5, "src": 2, "src2": 7, "imm": "0x7c170030", "imm2": "0xa3102916", "rot": 5, "bit": 6, "mask": 1},
{"i": 152, "op": "mul", "dst": 0, "src": 7, "src2": 5, "imm": "0x16d009af", "imm2": "0x2a89533b", "rot": 26, "bit": 8, "mask": 4},
{"i": 153, "op": "rotl", "dst": 4, "src": 5, "src2": 0, "imm": "0x6e54eb71", "imm2": "0xe2d27256", "rot": 16, "bit": 18, "mask": 1},
{"i": 154, "op": "rotr", "dst": 6, "src": 4, "src2": 5, "imm": "0x128f24a4", "imm2": "0x59888463", "rot": 15, "bit": 17, "mask": 16},
{"i": 155, "op": "xor", "dst": 6, "src": 1, "src2": 2, "imm": "0x34f276ff", "imm2": "0xef5086b4", "rot": 2, "bit": 10, "mask": 4},
{"i": 156, "op": "or", "dst": 6, "src": 3, "src2": 4, "imm": "0x16cc2c2f", "imm2": "0xfae5a09b", "rot": 12, "bit": 31, "mask": 4},
{"i": 157, "op": "add", "dst": 1, "src": 6, "src2": 7, "imm": "0x14878c5a", "imm2": "0x84c2a09d", "rot": 30, "bit": 19, "mask": 8},
{"i": 158, "op": "sub", "dst": 4, "src": 1, "src2": 0, "imm": "0xfe18e4b0", "imm2": "0x239df0cc", "rot": 14, "bit": 20, "mask": 2},
{"i": 159, "op": "mad", "dst": 4, "src": 6, "src2": 7, "imm": "0xbdfcc4e0", "imm2": "0x0870dee8", "rot": 1, "bit": 7, "mask": 8},
{"i": 160, "op": "mul", "dst": 1, "src": 5, "src2": 1, "imm": "0xc837b43d", "imm2": "0x027ac950", "rot": 4, "bit": 22, "mask": 16},
{"i": 161, "op": "or", "dst": 4, "src": 0, "src2": 2, "imm": "0x9c28f8f6", "imm2": "0x3db71fe6", "rot": 11, "bit": 6, "mask": 16},
{"i": 162, "op": "rotl", "dst": 7, "src": 3, "src2": 6, "imm": "0xb6e56f55", "imm2": "0x5fcbd27c", "rot": 21, "bit": 10, "mask": 1},
{"i": 163, "op": "sub", "dst": 0, "src": 1, "src2": 4, "imm": "0x8d95ae70", "imm2": "0xd4f10284", "rot": 15, "bit": 12, "mask": 8},
{"i": 164, "op": "mul", "dst": 1, "src": 0, "src2": 6, "imm": "0xa9bdf99e", "imm2": "0x1f3dceb6", "rot": 21, "bit": 20, "mask": 16},
{"i": 165, "op": "sub", "dst": 2, "src": 1, "src2": 4, "imm": "0xda245058", "imm2": "0x2038e7f6", "rot": 30, "bit": 11, "mask": 2},
{"i": 166, "op": "shfl", "dst": 0, "src": 6, "src2": 3, "imm": "0xbbc8eee8", "imm2": "0xa73472fb", "rot": 23, "bit": 4, "mask": 2},
{"i": 167, "op": "sub", "dst": 0, "src": 4, "src2": 2, "imm": "0x43540d69", "imm2": "0xb3f116ca", "rot": 28, "bit": 2, "mask": 2},
{"i": 168, "op": "rotr", "dst": 1, "src": 6, "src2": 6, "imm": "0x57cea752", "imm2": "0xc1136e71", "rot": 25, "bit": 3, "mask": 8},
{"i": 169, "op": "rotr", "dst": 7, "src": 1, "src2": 5, "imm": "0x4550a465", "imm2": "0x14362935", "rot": 25, "bit": 9, "mask": 16},
{"i": 170, "op": "mad", "dst": 3, "src": 1, "src2": 3, "imm": "0x8c961d80", "imm2": "0x7de2e1db", "rot": 10, "bit": 6, "mask": 4},
{"i": 171, "op": "shfl", "dst": 2, "src": 0, "src2": 0, "imm": "0xda047fca", "imm2": "0x459e17ab", "rot": 22, "bit": 26, "mask": 4},
{"i": 172, "op": "sub", "dst": 2, "src": 1, "src2": 2, "imm": "0xbcc341fd", "imm2": "0x17a3b229", "rot": 8, "bit": 25, "mask": 8},
{"i": 173, "op": "xor", "dst": 7, "src": 6, "src2": 7, "imm": "0x2302b424", "imm2": "0x78780982", "rot": 23, "bit": 7, "mask": 8},
{"i": 174, "op": "rotr", "dst": 4, "src": 1, "src2": 2, "imm": "0xcad6fb5f", "imm2": "0x1807628e", "rot": 8, "bit": 21, "mask": 4},
{"i": 175, "op": "shfl", "dst": 4, "src": 7, "src2": 7, "imm": "0xf5057e8e", "imm2": "0xd14d6cc5", "rot": 3, "bit": 12, "mask": 2},
{"i": 176, "op": "mul", "dst": 6, "src": 2, "src2": 1, "imm": "0xc2926356", "imm2": "0xae74dc32", "rot": 1, "bit": 21, "mask": 8},
{"i": 177, "op": "shfl", "dst": 0, "src": 2, "src2": 7, "imm": "0x8d94d98e", "imm2": "0x7bd6f4dc", "rot": 27, "bit": 14, "mask": 4},
{"i": 178, "op": "add", "dst": 6, "src": 4, "src2": 6, "imm": "0x4fa43965", "imm2": "0x509871e6", "rot": 28, "bit": 21, "mask": 4},
{"i": 179, "op": "xor", "dst": 0, "src": 6, "src2": 6, "imm": "0xe278fe6c", "imm2": "0x75d8a63f", "rot": 21, "bit": 15, "mask": 2},
{"i": 180, "op": "add", "dst": 5, "src": 7, "src2": 7, "imm": "0xfb3c4bd8", "imm2": "0xb95b8a53", "rot": 22, "bit": 0, "mask": 4},
{"i": 181, "op": "mul", "dst": 6, "src": 3, "src2": 2, "imm": "0x5a1347ec", "imm2": "0x3ccc5389", "rot": 27, "bit": 26, "mask": 4},
{"i": 182, "op": "rotl", "dst": 7, "src": 4, "src2": 7, "imm": "0xdc6db2f7", "imm2": "0x7cdff0c6", "rot": 1, "bit": 2, "mask": 4},
{"i": 183, "op": "add", "dst": 0, "src": 7, "src2": 7, "imm": "0xc448a197", "imm2": "0x7047b7cf", "rot": 24, "bit": 11, "mask": 8},
{"i": 184, "op": "mul", "dst": 3, "src": 5, "src2": 0, "imm": "0xa5c0e492", "imm2": "0xaf7d8a86", "rot": 11, "bit": 16, "mask": 8},
{"i": 185, "op": "shfl", "dst": 5, "src": 7, "src2": 1, "imm": "0x5965794b", "imm2": "0x36ddca6b", "rot": 7, "bit": 25, "mask": 1},
{"i": 186, "op": "mul", "dst": 0, "src": 2, "src2": 5, "imm": "0xdfdbcc0f", "imm2": "0xac7b7091", "rot": 11, "bit": 31, "mask": 8},
{"i": 187, "op": "mulhi", "dst": 4, "src": 7, "src2": 5, "imm": "0x3d8999e9", "imm2": "0xf1848db8", "rot": 15, "bit": 21, "mask": 8},
{"i": 188, "op": "mad", "dst": 2, "src": 7, "src2": 2, "imm": "0xda3af48c", "imm2": "0x8e33a97c", "rot": 6, "bit": 30, "mask": 4},
{"i": 189, "op": "sub", "dst": 7, "src": 2, "src2": 4, "imm": "0xa32da8c1", "imm2": "0xf7486d06", "rot": 25, "bit": 16, "mask": 4},
{"i": 190, "op": "add", "dst": 2, "src": 5, "src2": 1, "imm": "0x2fceaf49", "imm2": "0x5caccc5d", "rot": 24, "bit": 19, "mask": 2},
{"i": 191, "op": "shfl", "dst": 1, "src": 2, "src2": 2, "imm": "0x2a71f5e6", "imm2": "0xb9b51e89", "rot": 31, "bit": 25, "mask": 1},
{"i": 192, "op": "add", "dst": 7, "src": 1, "src2": 7, "imm": "0xba0cee62", "imm2": "0x857bb5fe", "rot": 4, "bit": 20, "mask": 2},
{"i": 193, "op": "rotl", "dst": 6, "src": 3, "src2": 3, "imm": "0xc7fe35cf", "imm2": "0x515c1201", "rot": 22, "bit": 7, "mask": 16},
{"i": 194, "op": "xor", "dst": 3, "src": 7, "src2": 4, "imm": "0xe496d1f1", "imm2": "0x11046e3e", "rot": 16, "bit": 9, "mask": 4},
{"i": 195, "op": "mul", "dst": 7, "src": 0, "src2": 2, "imm": "0xc7d5c4c2", "imm2": "0x077e940b", "rot": 23, "bit": 29, "mask": 4},
{"i": 196, "op": "xor", "dst": 3, "src": 5, "src2": 3, "imm": "0xbe6db14a", "imm2": "0xf736da5a", "rot": 19, "bit": 6, "mask": 2},
{"i": 197, "op": "add", "dst": 5, "src": 1, "src2": 5, "imm": "0x8519428c", "imm2": "0xeae84577", "rot": 18, "bit": 29, "mask": 1},
{"i": 198, "op": "mad", "dst": 7, "src": 4, "src2": 4, "imm": "0xb272a029", "imm2": "0x071f6a5c", "rot": 5, "bit": 28, "mask": 8},
{"i": 199, "op": "add", "dst": 3, "src": 6, "src2": 3, "imm": "0x468639d3", "imm2": "0x0bfdbfa1", "rot": 20, "bit": 8, "mask": 16},
{"i": 200, "op": "mad", "dst": 5, "src": 3, "src2": 4, "imm": "0xffba18b7", "imm2": "0xbdcf9e81", "rot": 13, "bit": 31, "mask": 8},
{"i": 201, "op": "shfl", "dst": 2, "src": 7, "src2": 7, "imm": "0xb7533191", "imm2": "0x8d58cc8c", "rot": 7, "bit": 29, "mask": 4},
{"i": 202, "op": "add", "dst": 7, "src": 4, "src2": 6, "imm": "0x18ec9a69", "imm2": "0x53ee8e50", "rot": 12, "bit": 23, "mask": 16},
{"i": 203, "op": "xor", "dst": 6, "src": 0, "src2": 5, "imm": "0x12011ff1", "imm2": "0xfbb57ac6", "rot": 16, "bit": 25, "mask": 16},
{"i": 204, "op": "add", "dst": 4, "src": 5, "src2": 3, "imm": "0x3e81485b", "imm2": "0xac63376e", "rot": 1, "bit": 4, "mask": 1},
{"i": 205, "op": "mad", "dst": 7, "src": 1, "src2": 4, "imm": "0xeb9eee5c", "imm2": "0x3f1daecd", "rot": 19, "bit": 29, "mask": 4},
{"i": 206, "op": "shfl", "dst": 1, "src": 5, "src2": 2, "imm": "0x4fc50d06", "imm2": "0xb5533b31", "rot": 22, "bit": 27, "mask": 16},
{"i": 207, "op": "mul", "dst": 4, "src": 0, "src2": 5, "imm": "0x202d6001", "imm2": "0x50388f85", "rot": 5, "bit": 18, "mask": 16},
{"i": 208, "op": "or", "dst": 1, "src": 6, "src2": 7, "imm": "0x8743c0d4", "imm2": "0x9df69539", "rot": 14, "bit": 16, "mask": 1},
{"i": 209, "op": "add", "dst": 6, "src": 4, "src2": 6, "imm": "0x66ccb75d", "imm2": "0x5b745519", "rot": 28, "bit": 28, "mask": 8},
{"i": 210, "op": "rotr", "dst": 1, "src": 3, "src2": 1, "imm": "0xe9933132", "imm2": "0x6702fe85", "rot": 16, "bit": 19, "mask": 1},
{"i": 211, "op": "sub", "dst": 5, "src": 4, "src2": 4, "imm": "0xfe04b942", "imm2": "0x40dd2ce8", "rot": 24, "bit": 8, "mask": 4},
{"i": 212, "op": "xor", "dst": 4, "src": 7, "src2": 7, "imm": "0x10827878", "imm2": "0xef0de8fc", "rot": 17, "bit": 11, "mask": 8},
{"i": 213, "op": "add", "dst": 1, "src": 7, "src2": 1, "imm": "0x0ce0553d", "imm2": "0x1c3dccf7", "rot": 26, "bit": 8, "mask": 1},
{"i": 214, "op": "mulhi", "dst": 0, "src": 1, "src2": 1, "imm": "0xa71d7581", "imm2": "0xe567c74c", "rot": 2, "bit": 29, "mask": 2},
{"i": 215, "op": "add", "dst": 3, "src": 6, "src2": 2, "imm": "0x96bfec89", "imm2": "0xfadc9205", "rot": 30, "bit": 9, "mask": 16},
{"i": 216, "op": "rotl", "dst": 0, "src": 2, "src2": 5, "imm": "0x947ca474", "imm2": "0xd2a1b650", "rot": 4, "bit": 1, "mask": 1},
{"i": 217, "op": "xor", "dst": 6, "src": 4, "src2": 5, "imm": "0x0b7617de", "imm2": "0xc6c0a9a2", "rot": 31, "bit": 15, "mask": 2},
{"i": 218, "op": "shfl", "dst": 6, "src": 7, "src2": 1, "imm": "0x738f0a8f", "imm2": "0xfc868560", "rot": 2, "bit": 20, "mask": 2},
{"i": 219, "op": "mad", "dst": 7, "src": 0, "src2": 1, "imm": "0x2960f2bc", "imm2": "0x7e9ef9b7", "rot": 21, "bit": 4, "mask": 16},
{"i": 220, "op": "mul", "dst": 4, "src": 0, "src2": 2, "imm": "0x6014d800", "imm2": "0xd19ac7d1", "rot": 1, "bit": 13, "mask": 1},
{"i": 221, "op": "mad", "dst": 2, "src": 6, "src2": 0, "imm": "0x4b558fc2", "imm2": "0x2d48bd3c", "rot": 18, "bit": 31, "mask": 2},
{"i": 222, "op": "xor", "dst": 5, "src": 4, "src2": 6, "imm": "0x0260f3ab", "imm2": "0x8fce22f1", "rot": 21, "bit": 26, "mask": 16},
{"i": 223, "op": "mul", "dst": 0, "src": 5, "src2": 0, "imm": "0xe07ab58f", "imm2": "0x5349a41d", "rot": 20, "bit": 3, "mask": 1},
{"i": 224, "op": "add", "dst": 2, "src": 0, "src2": 5, "imm": "0x61495681", "imm2": "0x66b15c04", "rot": 24, "bit": 9, "mask": 16},
{"i": 225, "op": "rotr", "dst": 3, "src": 2, "src2": 5, "imm": "0x291268bd", "imm2": "0xf046aa79", "rot": 27, "bit": 15, "mask": 4},
{"i": 226, "op": "mad", "dst": 2, "src": 3, "src2": 3, "imm": "0xfd6dfa6c", "imm2": "0x2c0db93d", "rot": 29, "bit": 2, "mask": 16},
{"i": 227, "op": "xor", "dst": 6, "src": 0, "src2": 4, "imm": "0xd4ae1ee2", "imm2": "0xe6a67972", "rot": 26, "bit": 1, "mask": 4},
{"i": 228, "op": "rotl", "dst": 4, "src": 7, "src2": 3, "imm": "0xf6627a9d", "imm2": "0x99078da3", "rot": 30, "bit": 5, "mask": 16},
{"i": 229, "op": "add", "dst": 2, "src": 3, "src2": 7, "imm": "0x70d05c34", "imm2": "0x25bc17c2", "rot": 20, "bit": 29, "mask": 4},
{"i": 230, "op": "rotr", "dst": 1, "src": 5, "src2": 2, "imm": "0xef874bea", "imm2": "0x3bfbfecc", "rot": 17, "bit": 2, "mask": 2},
{"i": 231, "op": "add", "dst": 1, "src": 4, "src2": 5, "imm": "0x32a28384", "imm2": "0x2d8728f3", "rot": 1, "bit": 8, "mask": 8},
{"i": 232, "op": "rotl", "dst": 1, "src": 6, "src2": 3, "imm": "0x877c54ab", "imm2": "0x1d5e5014", "rot": 6, "bit": 24, "mask": 2},
{"i": 233, "op": "add", "dst": 2, "src": 0, "src2": 5, "imm": "0xcf0949f1", "imm2": "0x61e81fe6", "rot": 2, "bit": 21, "mask": 16},
{"i": 234, "op": "rotl", "dst": 4, "src": 5, "src2": 5, "imm": "0x5017c32c", "imm2": "0x1ab34dfe", "rot": 2, "bit": 1, "mask": 2},
{"i": 235, "op": "sub", "dst": 2, "src": 6, "src2": 2, "imm": "0xadbdb811", "imm2": "0x2c2d4f2b", "rot": 30, "bit": 29, "mask": 8},
{"i": 236, "op": "or", "dst": 6, "src": 3, "src2": 4, "imm": "0x700e204f", "imm2": "0x27294e7b", "rot": 13, "bit": 0, "mask": 4},
{"i": 237, "op": "shfl", "dst": 7, "src": 5, "src2": 4, "imm": "0xc39f71ba", "imm2": "0x9a90263d", "rot": 28, "bit": 9, "mask": 4},
{"i": 238, "op": "xor", "dst": 1, "src": 3, "src2": 6, "imm": "0x9f84a5b3", "imm2": "0x95691d15", "rot": 14, "bit": 22, "mask": 1},
{"i": 239, "op": "rotl", "dst": 2, "src": 4, "src2": 0, "imm": "0xa3b79ed5", "imm2": "0x05f568bc", "rot": 1, "bit": 15, "mask": 4},
{"i": 240, "op": "rotr", "dst": 5, "src": 2, "src2": 7, "imm": "0xaaabbb0d", "imm2": "0x71ce4cd6", "rot": 2, "bit": 26, "mask": 4},
{"i": 241, "op": "rotl", "dst": 4, "src": 1, "src2": 7, "imm": "0x8146fbd3", "imm2": "0xde65fc31", "rot": 10, "bit": 9, "mask": 4},
{"i": 242, "op": "add", "dst": 6, "src": 2, "src2": 7, "imm": "0xe204fb50", "imm2": "0x0a67565d", "rot": 5, "bit": 16, "mask": 1},
{"i": 243, "op": "shfl", "dst": 1, "src": 3, "src2": 7, "imm": "0xcb7a3c1b", "imm2": "0x863f002c", "rot": 18, "bit": 1, "mask": 2},
{"i": 244, "op": "shfl", "dst": 0, "src": 1, "src2": 3, "imm": "0x92eb030c", "imm2": "0x3cff4c6d", "rot": 1, "bit": 22, "mask": 16},
{"i": 245, "op": "shfl", "dst": 5, "src": 7, "src2": 6, "imm": "0x046c969c", "imm2": "0xd21d7e76", "rot": 5, "bit": 31, "mask": 2},
{"i": 246, "op": "xor", "dst": 4, "src": 5, "src2": 1, "imm": "0x2e3015f0", "imm2": "0x1b305a5c", "rot": 22, "bit": 16, "mask": 8},
{"i": 247, "op": "xor", "dst": 7, "src": 2, "src2": 0, "imm": "0x8238cf71", "imm2": "0xb7874d94", "rot": 9, "bit": 1, "mask": 2},
{"i": 248, "op": "shfl", "dst": 0, "src": 6, "src2": 7, "imm": "0x437cbc99", "imm2": "0x7fb41fed", "rot": 17, "bit": 1, "mask": 16},
{"i": 249, "op": "mulhi", "dst": 1, "src": 3, "src2": 2, "imm": "0xeb293d88", "imm2": "0xb92f5968", "rot": 3, "bit": 20, "mask": 2},
{"i": 250, "op": "add", "dst": 7, "src": 5, "src2": 3, "imm": "0x3d6daffd", "imm2": "0x5a79fd6f", "rot": 6, "bit": 31, "mask": 2},
{"i": 251, "op": "add", "dst": 6, "src": 0, "src2": 3, "imm": "0x84334aea", "imm2": "0x6beb28c2", "rot": 27, "bit": 28, "mask": 2},
{"i": 252, "op": "sub", "dst": 7, "src": 1, "src2": 6, "imm": "0xe6177638", "imm2": "0x814d3fc5", "rot": 25, "bit": 2, "mask": 2},
{"i": 253, "op": "rotr", "dst": 5, "src": 2, "src2": 7, "imm": "0x95560bac", "imm2": "0xbf8b01a9", "rot": 14, "bit": 5, "mask": 16},
{"i": 254, "op": "mulhi", "dst": 2, "src": 5, "src2": 1, "imm": "0x318956e5", "imm2": "0x7c20b373", "rot": 13, "bit": 30, "mask": 2},
{"i": 255, "op": "mul", "dst": 6, "src": 3, "src2": 3, "imm": "0xe1998e49", "imm2": "0x59353e5a", "rot": 15, "bit": 2, "mask": 4}
]},
"instructions": [
{"i": 0, "op": "sub", "dst": 6, "src": 3, "src2": 3, "imm": "0x77dfbe60", "imm2": "0x404c8b9c", "rot": 24, "bit": 8, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 1, "op": "or", "dst": 2, "src": 6, "src2": 7, "imm": "0xb2e79058", "imm2": "0xc2485073", "rot": 21, "bit": 23, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 2, "op": "add", "dst": 4, "src": 0, "src2": 6, "imm": "0x1a579b38", "imm2": "0x7731324b", "rot": 6, "bit": 4, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 3, "op": "load", "dst": 3, "src": 6, "src2": 0, "imm": "0x683f5bc5", "imm2": "0xc8a218fe", "rot": 25, "bit": 4, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 4, "op": "shfl", "dst": 0, "src": 6, "src2": 4, "imm": "0x4d7cdcbb", "imm2": "0x1a5bf1a4", "rot": 8, "bit": 28, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 5, "op": "xor", "dst": 4, "src": 1, "src2": 0, "imm": "0xa2b86827", "imm2": "0x94a44439", "rot": 21, "bit": 22, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 6, "op": "mad", "dst": 3, "src": 5, "src2": 3, "imm": "0x58a4ea3f", "imm2": "0x467879fa", "rot": 6, "bit": 31, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 7, "op": "mul", "dst": 1, "src": 6, "src2": 1, "imm": "0x31fcc19c", "imm2": "0x176cb88f", "rot": 1, "bit": 11, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 8, "op": "load", "dst": 2, "src": 0, "src2": 3, "imm": "0x89b12386", "imm2": "0x661ed5e5", "rot": 11, "bit": 13, "mask": 2, "width": 1, "win": 2, "off": 2},
{"i": 9, "op": "xor", "dst": 0, "src": 4, "src2": 1, "imm": "0x69150105", "imm2": "0x8e1f04b9", "rot": 3, "bit": 30, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 10, "op": "add", "dst": 0, "src": 7, "src2": 5, "imm": "0xee083919", "imm2": "0xb10fcef8", "rot": 9, "bit": 30, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 11, "op": "mulhi", "dst": 0, "src": 2, "src2": 3, "imm": "0x99ac7c83", "imm2": "0x34eb2845", "rot": 18, "bit": 12, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 12, "op": "xor", "dst": 5, "src": 4, "src2": 0, "imm": "0x97cfc887", "imm2": "0x0458f053", "rot": 28, "bit": 24, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 13, "op": "mad", "dst": 7, "src": 3, "src2": 0, "imm": "0x503e87b4", "imm2": "0x2c5d2617", "rot": 3, "bit": 15, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 14, "op": "load", "dst": 3, "src": 4, "src2": 4, "imm": "0xfae98035", "imm2": "0x6c127281", "rot": 19, "bit": 29, "mask": 4, "width": 1, "win": 1, "off": 0},
{"i": 15, "op": "load", "dst": 2, "src": 5, "src2": 2, "imm": "0x67601ed5", "imm2": "0x3c546ed1", "rot": 28, "bit": 28, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 16, "op": "shfl", "dst": 7, "src": 5, "src2": 6, "imm": "0x7789be79", "imm2": "0x87496b8e", "rot": 22, "bit": 2, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 17, "op": "mul", "dst": 7, "src": 6, "src2": 2, "imm": "0x559099bb", "imm2": "0x6a72e1a6", "rot": 1, "bit": 11, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 18, "op": "mad", "dst": 2, "src": 3, "src2": 1, "imm": "0x4f5bae5e", "imm2": "0x4c8c5762", "rot": 27, "bit": 31, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 19, "op": "rotr", "dst": 5, "src": 2, "src2": 4, "imm": "0x5dc5e6bf", "imm2": "0x36bc207d", "rot": 9, "bit": 4, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 20, "op": "load", "dst": 5, "src": 3, "src2": 7, "imm": "0x4a278885", "imm2": "0xcd1044b0", "rot": 30, "bit": 0, "mask": 4, "width": 1, "win": 1, "off": 1},
{"i": 21, "op": "shfl", "dst": 6, "src": 4, "src2": 3, "imm": "0x48b94ffc", "imm2": "0xd8e14d80", "rot": 2, "bit": 31, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 22, "op": "mulhi", "dst": 1, "src": 2, "src2": 7, "imm": "0x47bd8492", "imm2": "0x28fa8d0b", "rot": 19, "bit": 26, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 23, "op": "xor", "dst": 6, "src": 5, "src2": 1, "imm": "0x4ed6baf4", "imm2": "0xa8c31f74", "rot": 30, "bit": 2, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 24, "op": "mad", "dst": 5, "src": 2, "src2": 0, "imm": "0x6174747d", "imm2": "0xfc912524", "rot": 18, "bit": 29, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 25, "op": "mad", "dst": 6, "src": 3, "src2": 7, "imm": "0x113a6442", "imm2": "0xbe52ec57", "rot": 15, "bit": 15, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 26, "op": "load", "dst": 6, "src": 5, "src2": 1, "imm": "0x9d3eb1cc", "imm2": "0x375c73bb", "rot": 29, "bit": 7, "mask": 8, "width": 1, "win": 1, "off": 1},
{"i": 27, "op": "mad", "dst": 1, "src": 5, "src2": 5, "imm": "0x150088c3", "imm2": "0x7f17e789", "rot": 8, "bit": 25, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 28, "op": "load", "dst": 0, "src": 6, "src2": 3, "imm": "0x4b412223", "imm2": "0x9192ceeb", "rot": 16, "bit": 28, "mask": 1, "width": 1, "win": 1, "off": 1},
{"i": 29, "op": "add", "dst": 7, "src": 4, "src2": 6, "imm": "0xd5e6c37c", "imm2": "0x0b1f02c7", "rot": 2, "bit": 7, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 30, "op": "add", "dst": 7, "src": 0, "src2": 6, "imm": "0x4e45ba51", "imm2": "0x2d6e8bd1", "rot": 7, "bit": 0, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 31, "op": "xor", "dst": 0, "src": 2, "src2": 7, "imm": "0xef601d89", "imm2": "0x805402db", "rot": 7, "bit": 28, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 32, "op": "add", "dst": 6, "src": 2, "src2": 3, "imm": "0x1c91b2a1", "imm2": "0x351422d2", "rot": 1, "bit": 13, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 33, "op": "mulhi", "dst": 4, "src": 5, "src2": 7, "imm": "0x7cf21567", "imm2": "0x1075ee0d", "rot": 26, "bit": 17, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 34, "op": "rotr", "dst": 3, "src": 5, "src2": 6, "imm": "0x60d8f191", "imm2": "0x40523707", "rot": 3, "bit": 10, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 35, "op": "load", "dst": 3, "src": 0, "src2": 5, "imm": "0xb1dc3f6e", "imm2": "0x74440531", "rot": 21, "bit": 10, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 36, "op": "rotl", "dst": 4, "src": 6, "src2": 4, "imm": "0xf0d17569", "imm2": "0xd281fd01", "rot": 13, "bit": 4, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 37, "op": "add", "dst": 6, "src": 1, "src2": 6, "imm": "0x3f2970f7", "imm2": "0x64a28251", "rot": 4, "bit": 7, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 38, "op": "mul", "dst": 2, "src": 0, "src2": 3, "imm": "0x6463770c", "imm2": "0xb85d603f", "rot": 27, "bit": 1, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 39, "op": "add", "dst": 3, "src": 4, "src2": 3, "imm": "0x3b7b3317", "imm2": "0x7378c955", "rot": 28, "bit": 15, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 40, "op": "load", "dst": 0, "src": 7, "src2": 3, "imm": "0xd0fb5982", "imm2": "0x2e4d31a8", "rot": 25, "bit": 27, "mask": 4, "width": 1, "win": 2, "off": 3},
{"i": 41, "op": "xor", "dst": 6, "src": 5, "src2": 0, "imm": "0xe96a9e6e", "imm2": "0x8d73cc3e", "rot": 17, "bit": 11, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 42, "op": "add", "dst": 3, "src": 0, "src2": 6, "imm": "0x6678b059", "imm2": "0xb31f6a64", "rot": 11, "bit": 2, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 43, "op": "load", "dst": 7, "src": 3, "src2": 2, "imm": "0x65a841c5", "imm2": "0x6aae1403", "rot": 9, "bit": 9, "mask": 1, "width": 1, "win": 1, "off": 0},
{"i": 44, "op": "shfl", "dst": 3, "src": 7, "src2": 6, "imm": "0x650ba38d", "imm2": "0x07218508", "rot": 12, "bit": 13, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 45, "op": "mulhi", "dst": 1, "src": 5, "src2": 7, "imm": "0x18736788", "imm2": "0x9f820eed", "rot": 13, "bit": 29, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 46, "op": "rotl", "dst": 0, "src": 1, "src2": 5, "imm": "0x48deba75", "imm2": "0xefc508fa", "rot": 6, "bit": 6, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 47, "op": "load", "dst": 1, "src": 7, "src2": 4, "imm": "0x28a04384", "imm2": "0xc67cb0ac", "rot": 5, "bit": 5, "mask": 2, "width": 1, "win": 1, "off": 1},
{"i": 48, "op": "add", "dst": 4, "src": 0, "src2": 5, "imm": "0x43095946", "imm2": "0x5feccee6", "rot": 22, "bit": 17, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 49, "op": "load", "dst": 2, "src": 0, "src2": 2, "imm": "0x3e78cdd8", "imm2": "0x7c170311", "rot": 10, "bit": 5, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 50, "op": "sub", "dst": 2, "src": 3, "src2": 7, "imm": "0xaff0cdb2", "imm2": "0xde0b7ee0", "rot": 21, "bit": 11, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 51, "op": "add", "dst": 7, "src": 2, "src2": 1, "imm": "0xf45ecdf8", "imm2": "0x26e2b582", "rot": 28, "bit": 24, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 52, "op": "load", "dst": 5, "src": 7, "src2": 7, "imm": "0x67353e69", "imm2": "0xb2d56082", "rot": 27, "bit": 14, "mask": 2, "width": 1, "win": 2, "off": 3},
{"i": 53, "op": "load", "dst": 6, "src": 3, "src2": 1, "imm": "0x5a58c787", "imm2": "0x53a0467b", "rot": 6, "bit": 16, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 54, "op": "shfl", "dst": 7, "src": 5, "src2": 4, "imm": "0x3c237ad5", "imm2": "0xf2681187", "rot": 14, "bit": 26, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 55, "op": "rotr", "dst": 1, "src": 2, "src2": 3, "imm": "0x1c25c077", "imm2": "0x125de1ff", "rot": 13, "bit": 13, "mask": 2, "width": 1, "win": 0, "off": 0},
{"i": 56, "op": "shfl", "dst": 0, "src": 5, "src2": 0, "imm": "0x6465afe2", "imm2": "0x23b41d56", "rot": 31, "bit": 28, "mask": 16, "width": 1, "win": 0, "off": 0},
{"i": 57, "op": "rotl", "dst": 5, "src": 0, "src2": 2, "imm": "0xa427ff6e", "imm2": "0x014a0deb", "rot": 12, "bit": 28, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 58, "op": "add", "dst": 5, "src": 3, "src2": 7, "imm": "0x6b4f35e8", "imm2": "0x473b1718", "rot": 25, "bit": 4, "mask": 8, "width": 1, "win": 0, "off": 0},
{"i": 59, "op": "rotr", "dst": 4, "src": 1, "src2": 3, "imm": "0x66cc96ef", "imm2": "0xcb8c8e56", "rot": 7, "bit": 6, "mask": 4, "width": 1, "win": 0, "off": 0},
{"i": 60, "op": "shfl", "dst": 4, "src": 5, "src2": 5, "imm": "0x461bbf9f", "imm2": "0xc1ffd350", "rot": 19, "bit": 2, "mask": 1, "width": 1, "win": 0, "off": 0},
{"i": 61, "op": "load", "dst": 7, "src": 0, "src2": 1, "imm": "0x4d893571", "imm2": "0x5583e644", "rot": 19, "bit": 19, "mask": 8, "width": 1, "win": 1, "off": 1},
{"i": 62, "op": "load", "dst": 5, "src": 7, "src2": 3, "imm": "0x7a9049b4", "imm2": "0xbb7ebdf8", "rot": 15, "bit": 11, "mask": 8, "width": 1, "win": 1, "off": 1},
{"i": 63, "op": "mul", "dst": 4, "src": 1, "src2": 7, "imm": "0xe7ae8a83", "imm2": "0x1ded64e8", "rot": 21, "bit": 1, "mask": 8, "width": 1, "win": 0, "off": 0}
]
}

File diff suppressed because it is too large Load diff

View file

@ -12,8 +12,11 @@ dir=/srv/builds/_adv-adv-accept # OUTSIDE the worktree mirror (correction 18:5
mkdir -p "$dir"
log="$dir/$tag.log"
[ -x "$bin" ] || { echo "no binary at $bin" >&2; exit 1; }
setsid nice -n 10 taskset -c 8-95 "$bin" "$@" > "$log" 2>&1 < /dev/null &
# the per-box sweep lock (coordinator, 19:55 BST, 7 October 2026): one bounded run per box at a time, the rest wait in
# order; the lock holder is the recorded pid (kill -- -<pgid> ends the wait or the run)
printf -v cmd '%q ' nice -n 10 taskset -c 8-95 "$bin" "$@"
setsid flock /srv/builds/_adv/locks/sweep.lock -c "$cmd > $(printf '%q' "$log") 2>&1" < /dev/null > /dev/null 2>&1 &
pid=$!
pgid=$(ps -o pgid= -p "$pid" 2>/dev/null | tr -d ' '); [ -n "$pgid" ] || pgid="$pid"
echo "$pid" > "$log.pid"; echo "$pgid" > "$log.pgid"
echo "$(date -u +%FT%TZ) started pid $pid pgid $pgid nice 10 cores 8-95; log $log" | tee -a "$log.start"
echo "$(date -u +%FT%TZ) queued pid $pid pgid $pgid behind the per-box sweep lock; nice 10 cores 8-95; log $log" | tee -a "$log.start"